{
  "publisher": "Coval",
  "board": "https://benchmarks.coval.ai/tts",
  "harness": "https://github.com/coval-ai/benchmarks",
  "methodology": "https://github.com/coval-ai/benchmarks/blob/main/docs/methodology.md",
  "provenance": "SpeechifyAI's reading of Coval's public results API (/v2/results), the per-request rows behind benchmarks.coval.ai, produced by Coval's open-source harness, whose workers run in us-east-1 (harness README). Every TTS model Coval's /v1/providers lists as enabled, TTFA rows only, with the API's public defaults (successful evaluations, default variant), captured from 2026-09-23T19:05:00Z (inclusive) to 2026-09-24T19:05:00Z (exclusive), every page followed to the end. Read on 25 Sep 2026, inside the API's rolling retention. The bounds differ from coval-tts-2026-09-24.json, whose window24h.end records its newest Simba row (2026-09-24T19:02:13Z) rather than a query bound, but the Simba measurements are the same: Coval's Simba runs land at :01 to :04 and :31 to :33 past the hour, so both windows hold the same Simba rows, the oldest at 23 Sep 19:31:13 UTC and the newest at 24 Sep 19:01:47 UTC, which is why the medians and counts (106.3 ms over 150, 102.2 ms over 152) match. Starting this window at 19:02 on 23 Sep instead would add one more run.",
  "metric": "TTFA = (first audio chunk arrival - synthesis start) + leading silence inside the stream before the first audible sample, in milliseconds, as Coval's methodology defines it. medianTtfa is the median over the window; runs is the number of TTFA measurements in it, one per synthesized item, the same count coval-tts-2026-09-24.json calls runs.",
  "capturedOn": "2026-09-25",
  "date": "24 Sep 2026",
  "window": {"since": "2026-09-23T19:05:00Z", "until": "2026-09-24T19:05:00Z", "label": "24 h to 24 Sep"},
  "notes": "Baseten's qwen3-tts-1.7b is left out: its 30 rows in the window all fall in one three-minute burst at 04:21 UTC, while every other model's rows spread across the whole day. Deepgram's two models ran about 90 times in the window rather than about 150, and Deepdub's last run in it was at 15:31 UTC. Coval's own board may show a different window.",
  "models": [
    {"provider": "Gradium", "model": "gradium-tts-beta", "medianTtfa": 48.3, "runs": 152},
    {"provider": "Gradium", "model": "gradium-tts-beta-202609", "medianTtfa": 48.4, "runs": 151},
    {"provider": "Fluxions", "model": "vui", "medianTtfa": 51.4, "runs": 152},
    {"provider": "Inworld", "model": "inworld-tts-2-flash", "medianTtfa": 62.0, "runs": 152},
    {"provider": "Nari", "model": "qwen3-tts-fast", "medianTtfa": 63.6, "runs": 152},
    {"provider": "SpeechifyAI", "model": "simba-3.0", "medianTtfa": 102.2, "runs": 152, "ours": true},
    {"provider": "SpeechifyAI", "model": "simba-3.2", "medianTtfa": 106.3, "runs": 150, "ours": true},
    {"provider": "Palabra", "model": "palabra-tts-v1", "medianTtfa": 114.2, "runs": 137},
    {"provider": "Inworld", "model": "inworld-tts-2", "medianTtfa": 155.7, "runs": 152},
    {"provider": "ElevenLabs", "model": "eleven_flash_v2_5", "medianTtfa": 185.0, "runs": 152},
    {"provider": "Deepgram", "model": "flux-haley-en", "medianTtfa": 209.8, "runs": 84},
    {"provider": "Soniox", "model": "tts-rt-v1", "medianTtfa": 249.4, "runs": 140},
    {"provider": "Soniox", "model": "tts-rt-v2", "medianTtfa": 257.0, "runs": 140},
    {"provider": "Rime", "model": "mistv3", "medianTtfa": 260.8, "runs": 152},
    {"provider": "Deepdub", "model": "dd-etts-3.3", "medianTtfa": 263.5, "runs": 129},
    {"provider": "Deepgram", "model": "aura-2-thalia-en", "medianTtfa": 265.5, "runs": 93},
    {"provider": "Cartesia", "model": "sonic-3.5", "medianTtfa": 280.2, "runs": 152},
    {"provider": "Fish Audio", "model": "s2.1-pro", "medianTtfa": 307.7, "runs": 152},
    {"provider": "Rime", "model": "coda", "medianTtfa": 308.3, "runs": 152},
    {"provider": "Cartesia", "model": "sonic-3.6", "medianTtfa": 310.7, "runs": 159},
    {"provider": "ElevenLabs", "model": "eleven_v3_conversational", "medianTtfa": 325.7, "runs": 152},
    {"provider": "Smallest AI", "model": "lightning_v3.1_pro", "medianTtfa": 337.1, "runs": 152},
    {"provider": "Fish Audio", "model": "s1", "medianTtfa": 371.6, "runs": 152},
    {"provider": "xAI", "model": "grok-tts", "medianTtfa": 394.2, "runs": 152},
    {"provider": "Cloudflare", "model": "aura-2-en", "medianTtfa": 537.4, "runs": 152},
    {"provider": "Murf AI", "model": "falcon-2", "medianTtfa": 547.3, "runs": 152},
    {"provider": "Google", "model": "chirp-3-hd", "medianTtfa": 556.5, "runs": 152},
    {"provider": "Alibaba", "model": "qwen3-tts-flash-realtime", "medianTtfa": 606.9, "runs": 150},
    {"provider": "OpenAI", "model": "gpt-4o-mini-tts", "medianTtfa": 689.4, "runs": 152},
    {"provider": "Fish Audio", "model": "s2.1-pro-free", "medianTtfa": 892.0, "runs": 152}
  ]
}
