{
  "site": {
    "measured": "2026-07-03",
    "run": "run 07-03",
    "scope": "110 models · 42 integrations",
    "host": "benchmarks.speko.ai",
    "tagline": "Independent, reproducible voice-AI benchmarks: STT, TTS, LLM, S2S, and true cost-per-solve across the whole call.",
    "blurb": "Independent, reproducible voice-AI benchmarks: STT, TTS, LLM, S2S, turn-taking, and true cost-per-solve across the whole call. Published numbers are measured, never marketed."
  },
  "boards": [
    {
      "id": "stt",
      "navLabel": "Speech-to-Text",
      "identityLabel": "Provider / Model",
      "title": "Speech-to-Text",
      "subtitle": "Word error rate, latency and cost for every speech-to-text model we measure.",
      "live": true,
      "meta": "English · read speech (FLEURS, n=50) batch 2026-07-03 / streaming 2026-07-21 · spontaneous conversation, 21 speakers / 11 accents (EdAcc, n=99, real-time delivery) 2026-08-28",
      "measured": "2026-07-03",
      "headline": [
        "endpoint",
        "wer",
        "cost"
      ],
      "scatter": {
        "x": "endpoint",
        "y": "wer",
        "cornerLabel": "fast + accurate"
      },
      "columns": [
        {
          "id": "wer",
          "label": "WER · batch",
          "dir": "lower",
          "info": "Word Error Rate on FLEURS read English (n=50), batch API. Read speech, not studio-clean audio. Lower is better."
        },
        {
          "id": "werStream",
          "label": "WER · stream",
          "dir": "lower",
          "info": "WER on the same clips through the live streaming socket — the path a voice agent runs on. Lower is better."
        },
        {
          "id": "werConv",
          "label": "WER · conversational",
          "dir": "lower",
          "info": "WER on 98 clips of SPONTANEOUS conversation from 21 speakers and 11 self-reported accents (EdAcc), delivered to the live socket at real-time pace -- 100ms frames, 100ms apart -- because a phone call cannot deliver audio faster than it is spoken. WHAT THE CORPUS IS: accented spontaneous speech with false starts, self-corrections and overlapping repair, and half of it band-limited -- 49 of 98 clips roll off below 3.5 kHz, which is telephone bandwidth. WHAT IT IS NOT: it is not noisy and it is not distorted. The quietest clip measures 17.4 dB SNR, the median 45.6 dB, nothing falls below 15 dB, and no clip shows any sample clipping, so nothing here tests background noise, a buzzing microphone, or cross-talk. Those conditions need their own arms and do not exist on this board yet. Accent coverage is also uneven: Mainstream US, Indian and Southern British are 86 of the 98 clips and the other eight accents carry one or two each. References audited: 22 of 120 clips removed -- 10 where a <DTMF> marker covers untranscribed speech, 9 the models collectively contradict, 1 session scaffolding, 1 whose pooled contribution is pathological. A tail-overrun test then removed 1 clip whose segment contains speech the transcript does not cover -- 15 of 17 engines emit trailing tokens past the reference and 71% agree on the first, and it is the only clip of 99 that trips the test. A boundary audit trimmed 5 references where the EdAcc segment does not contain the words the transcript claims -- deletions only, never additions. RANKING: the ordering is not a ranking. Paired per-clip sign tests put the top six mutually indistinguishable, and each row records the peers it cannot be separated from. Lower is better."
        },
        {
          "id": "endpoint",
          "label": "Finalize",
          "dir": "lower",
          "info": "End-of-speech to final transcript, p50 (us-east4, n=30). Whisker spans p50→p90."
        },
        {
          "id": "ttft",
          "label": "Time to first token",
          "dir": "lower",
          "info": "First partial word, p50 (us-east4). Whisker spans p50→p90."
        },
        {
          "id": "cost",
          "label": "Cost / min",
          "dir": "lower",
          "info": "$ per minute of audio, vendor list, streaming path. Pay-as-you-go where offered, otherwise the cheapest paid plan. Lower = cheaper; \"—\" where a vendor publishes no per-minute rate."
        }
      ],
      "rows": [
        {
          "provider": "Smallest AI",
          "model": "Pulse",
          "id": "smallest:pulse",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "flag": "WER inflated by inverse-text-normalization — spelled-out numbers score as errors.",
          "rec": [
            {
              "label": "Streaming"
            },
            {
              "label": "Real-time phone"
            }
          ],
          "cells": {
            "werConv": {
              "v": "11.0%",
              "s": 11,
              "tone": "good",
              "note": "The live socket runs `pulse`, not `pulse-pro`."
            },
            "wer": {
              "v": "5.1%",
              "s": 5.1,
              "tone": "warn"
            },
            "werStream": {
              "v": "8.0%",
              "s": 8,
              "tone": "warn"
            },
            "endpoint": {
              "v": "178ms",
              "s": 178,
              "lo": 178,
              "hi": 196,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "good"
            },
            "ttft": {
              "v": "1.39s",
              "s": 1394,
              "lo": 1394,
              "hi": 2395,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "~$0.005",
              "s": 0.005,
              "note": "Streaming Pulse. The vendor never publishes an exact rate, and its own pricing page contradicts this model page with ~$0.009/min."
            }
          }
        },
        {
          "provider": "Deepgram",
          "model": "Nova-3",
          "id": "deepgram:nova-3",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "rec": [
            {
              "label": "Real-time phone"
            },
            {
              "label": "Streaming"
            }
          ],
          "cells": {
            "werConv": {
              "v": "15.1%",
              "s": 15.1,
              "tone": "warn",
              "note": "Mishears rather than drops: 7.9% substitutions against 5.2% deletions."
            },
            "wer": {
              "v": "12.0%",
              "s": 11.96,
              "tone": "bad",
              "ci": "n=50 · FLEURS clean read-English",
              "note": "Corrected 2026-09-02 from a published 9.8% that has no reproducible source. 11.96% is the mean per-clip WER of the committed batch artifact results/anchor-nova3-20260808.json (run of 2026-08-08), re-confirmed by rescoring that artifact with the current text normalizer — 11.96% as stored and 11.96% rescored."
            },
            "werStream": {
              "v": "12.9%",
              "s": 12.9,
              "tone": "bad"
            },
            "endpoint": {
              "v": "106ms",
              "s": 106,
              "lo": 106,
              "hi": 138,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "good"
            },
            "ttft": {
              "v": "1.12s",
              "s": 1124,
              "lo": 1124,
              "hi": 1582,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "$0.0048",
              "s": 0.0048,
              "note": "Streaming pay-as-you-go. Vendor flags it as a limited-time promotional rate; pre-recorded is $0.0043."
            }
          }
        },
        {
          "provider": "Deepgram",
          "model": "Flux",
          "id": "deepgram:flux-general-en",
          "measured": "2026-08-04",
          "status": {
            "label": "GOOD",
            "tone": "good"
          },
          "flag": "Finalize includes Flux’s own turn decision — no forced finalize, no batch endpoint.",
          "rec": [
            {
              "label": "Real-time phone"
            },
            {
              "label": "Streaming"
            }
          ],
          "cells": {
            "werConv": {
              "v": "17.8%",
              "s": 17.8,
              "tone": "warn",
              "note": "Flux never emits EndOfTurn when a clip ends as the speaker stops; the transcript sits in the last Update, and ours discarded it — 52 of 70 clips came back empty until it was flushed on close."
            },
            "wer": {
              "v": "—",
              "s": null
            },
            "werStream": {
              "v": "6.6%",
              "s": 6.6,
              "tone": "warn"
            },
            "endpoint": {
              "v": "406ms",
              "s": 406,
              "lo": 406,
              "hi": 1193,
              "ci": "p50–p90 · us-east4 · n=30 · 2026-08-04",
              "tone": "warn"
            },
            "ttft": {
              "v": "1.11s",
              "s": 1109,
              "lo": 1109,
              "hi": 1590,
              "ci": "p50–p90 · us-east4 · 2026-08-04",
              "tone": "good"
            },
            "cost": {
              "v": "$0.0065",
              "s": 0.0065,
              "note": "Flux English streaming pay-as-you-go. Vendor flags it as a limited-time promotional rate; flux-general-multi is $0.0078."
            }
          }
        },
        {
          "provider": "Cartesia",
          "model": "Ink-2",
          "id": "cartesia:ink-2",
          "status": {
            "label": "WEAK",
            "tone": "bad"
          },
          "flag": "Batch WER inflated by truncation; the streaming path it runs on scores better.",
          "rec": [],
          "cells": {
            "werConv": {
              "v": "12.5%",
              "s": 12.5,
              "tone": "good"
            },
            "wer": {
              "v": "11.0%",
              "s": 11,
              "tone": "bad"
            },
            "werStream": {
              "v": "9.9%",
              "s": 9.9,
              "tone": "warn"
            },
            "endpoint": {
              "v": "102ms",
              "s": 102,
              "lo": 102,
              "hi": 151,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "good"
            },
            "ttft": {
              "v": "1.16s",
              "s": 1161,
              "lo": 1161,
              "hi": 2063,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "$0.0090",
              "s": 0.009,
              "note": "Ink-2 bills 3 credits/sec on Pro, the cheapest paid plan; $0.0071 at Startup, $0.0067 at Scale. The 1 credit/sec rate is Ink-Whisper, a different model."
            }
          }
        },
        {
          "provider": "Alibaba",
          "model": "Qwen3-ASR",
          "id": "alibaba:qwen3-asr-flash",
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "flag": "Asia-hosted endpoint — p50 carries the network round-trip to Singapore.",
          "rec": [
            {
              "label": "Multilingual"
            },
            {
              "label": "Streaming"
            }
          ],
          "cells": {
            "werConv": {
              "v": "12.2%",
              "s": 12.2,
              "tone": "good",
              "note": "The live socket runs qwen3-asr-flash-realtime. UNUSABLE FOR A CALL: median 311s of wall clock per 7s clip on this corpus, up to 1123s -- accuracy is competitive, responsiveness is not. n=97."
            },
            "wer": {
              "v": "2.8%",
              "s": 2.8,
              "tone": "good"
            },
            "werStream": {
              "v": "4.0%",
              "s": 4,
              "tone": "good"
            },
            "endpoint": {
              "v": "424ms",
              "s": 424,
              "lo": 424,
              "hi": 606,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "warn"
            },
            "ttft": {
              "v": "6.60s",
              "s": 6596,
              "lo": 6596,
              "hi": 7314,
              "ci": "p50–p90 · us-east4",
              "tone": "bad"
            },
            "cost": {
              "v": "$0.0054",
              "s": 0.0054,
              "note": "qwen3-asr-flash-realtime, Singapore. The file/sync SKU is $0.0021."
            }
          }
        },
        {
          "provider": "Google",
          "model": "Chirp 3",
          "id": "google:chirp_3",
          "status": {
            "label": "GOOD",
            "tone": "good"
          },
          "rec": [
            {
              "label": "Multilingual"
            },
            {
              "label": "Global"
            }
          ],
          "cells": {
            "werConv": {
              "v": "17.3%",
              "s": 17.3,
              "tone": "warn",
              "ci": "95% CI 14.8–19.9 · n=99",
              "note": "Measured 2026-09-06, vendor-direct Speech V2 -- NOT the gateway path the finalize cell beside it uses. The date lives in this note rather than a row-level `measured` stamp on purpose: `measured` is per ROW, and this row's wer, werStream, endpoint and ttft cells are all older, so stamping the row would forward-date four measurements to buy a date for one."
            },
            "wer": {
              "v": "3.9%",
              "s": 3.9,
              "tone": "good"
            },
            "werStream": {
              "v": "7.4%",
              "s": 7.4,
              "tone": "warn"
            },
            "endpoint": {
              "v": "581ms",
              "s": 581,
              "lo": 581,
              "hi": 882,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "bad"
            },
            "ttft": {
              "v": "5.60s",
              "s": 5595,
              "lo": 5595,
              "hi": 6208,
              "ci": "p50–p90 · us-east4",
              "tone": "bad"
            },
            "cost": {
              "v": "$0.0160",
              "s": 0.016,
              "tone": "bad",
              "note": "Speech-to-Text V2 Standard, first tier — one rate for streaming, sync and batch. Dynamic Batch Recognition is $0.003."
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "Scribe v2 Realtime",
          "id": "elevenlabs:scribe_v2_realtime",
          "slug": "elevenlabs-scribe-v2",
          "measured": "2026-07-21",
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "flag": "Realtime metrics measured through the live gateway socket. The July 3 batch-labeled WER did not record the served model or transport, so it is withheld.",
          "rec": [
            {
              "label": "Balanced"
            },
            {
              "label": "Streaming"
            }
          ],
          "cells": {
            "werConv": {
              "v": "9.6%",
              "s": 9.6,
              "tone": "good",
              "note": "The live socket runs scribe_v2_realtime whatever you pin."
            },
            "wer": {
              "v": "-",
              "s": null
            },
            "werStream": {
              "v": "3.4%",
              "s": 3.4,
              "tone": "good"
            },
            "endpoint": {
              "v": "233ms",
              "s": 233,
              "lo": 233,
              "hi": 323,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "good"
            },
            "ttft": {
              "v": "2.18s",
              "s": 2184,
              "lo": 2184,
              "hi": 2273,
              "ci": "p50–p90 · us-east4",
              "tone": "warn"
            },
            "cost": {
              "v": "$0.0065",
              "s": 0.0065,
              "note": "Scribe v2 Realtime at $0.39/hr, flat across every tier. Batch Scribe v2 is $0.0037."
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "Scribe v2",
          "id": "elevenlabs:scribe_v2",
          "slug": "elevenlabs-scribe-v2-batch",
          "measured": "2026-06-03",
          "meta": "Historical batch measurement / FLEURS / n=50 / -16 LUFS / 2026-06-03",
          "flag": "Historical batch-labelled Scribe v2 measurement: 2.9% WER on FLEURS English (n=50, -16 LUFS), measured 2026-06-03. The exact execution route, gateway deployment, and upstream-served model version were not retained.",
          "batchOnly": true,
          "rec": [],
          "cells": {
            "wer": {
              "v": "2.9%",
              "s": 2.9,
              "tone": "good"
            },
            "werStream": {
              "v": "-",
              "s": null
            },
            "endpoint": {
              "v": "-",
              "s": null
            },
            "ttft": {
              "v": "-",
              "s": null
            },
            "cost": {
              "v": "$0.0037",
              "s": 0.0037
            }
          }
        },
        {
          "provider": "Microsoft",
          "model": "MAI-Transcribe-2",
          "id": "microsoft:MAI-Transcribe-2",
          "routable": false,
          "batchOnly": true,
          "measured": "2026-09-03",
          "meta": "Vendor-direct Azure Speech fast transcription (enhancedMode), eastus / FLEURS n=50 + EdAcc n=82 / 2026-09-03",
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "flag": "Batch API only — no streaming socket, so nothing on the live path is measured here.",
          "rec": [
            {
              "label": "Accuracy"
            },
            {
              "label": "Value"
            }
          ],
          "cells": {
            "werConv": {
              "v": "-",
              "s": null
            },
            "wer": {
              "v": "2.3%",
              "s": 2.3,
              "tone": "good",
              "note": "Same pass reproduced the Nova-3 anchor at its published 11.96%."
            },
            "werStream": {
              "v": "-",
              "s": null
            },
            "endpoint": {
              "v": "-",
              "s": null
            },
            "ttft": {
              "v": "-",
              "s": null
            },
            "cost": {
              "v": "$0.0017",
              "s": 0.0017,
              "note": "$0.10 per hour of audio, a launch price with no stated end date."
            }
          }
        },
        {
          "provider": "Inworld",
          "model": "Realtime STT-1",
          "id": "inworld:inworld-stt-1",
          "routable": false,
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "flag": "Finalizes eagerly — ~1 in 5 turns commit before speech ends.",
          "rec": [
            {
              "label": "Streaming"
            },
            {
              "label": "Value"
            }
          ],
          "cells": {
            "werConv": {
              "v": "11.8%",
              "s": 11.8,
              "tone": "good",
              "note": "KNOWN FAILURE MODE, EXCLUDED FROM THIS NUMBER. On one corpus clip its decode collapsed into a repetition loop — 29 reference words became 339 of \"co\", more word errors than it makes across the rest of the corpus combined. That clip is excluded for EVERY provider, so it is absent from all seventeen rows. A caller can trigger this, and an agent would answer the garbage."
            },
            "wer": {
              "v": "3.3%",
              "s": 3.3,
              "tone": "good"
            },
            "werStream": {
              "v": "3.6%",
              "s": 3.6,
              "tone": "good"
            },
            "endpoint": {
              "v": "139ms",
              "s": 139,
              "lo": 139,
              "hi": 158,
              "ci": "p50–p90 · us-east4 · n=50",
              "tone": "good"
            },
            "ttft": {
              "v": "1.21s",
              "s": 1209,
              "lo": 1209,
              "hi": 1807,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "$0.0025",
              "s": 0.0025,
              "tone": "good",
              "note": "On-demand $0.15/hr. The $0.10/hr headline needs a paid Creator subscription."
            }
          }
        },
        {
          "provider": "AssemblyAI",
          "model": "Universal-3.5 Pro",
          "id": "assemblyai:universal-3-5-pro",
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "rec": [
            {
              "label": "Streaming"
            },
            {
              "label": "Accuracy"
            }
          ],
          "cells": {
            "werConv": {
              "v": "10.4%",
              "s": 10.4,
              "tone": "good",
              "note": "Best substitution rate on the board at 4.1% -- it hears better than anything here and loses on emission, carrying 4.9% deletions."
            },
            "wer": {
              "v": "2.0%",
              "s": 2,
              "tone": "good"
            },
            "werStream": {
              "v": "2.0%",
              "s": 2,
              "tone": "good"
            },
            "endpoint": {
              "v": "66ms",
              "s": 66,
              "lo": 66,
              "hi": 81,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "good"
            },
            "ttft": {
              "v": "1.32s",
              "s": 1324,
              "lo": 1324,
              "hi": 2541,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "$0.0075",
              "s": 0.0075,
              "note": "Universal-3.5 Pro Realtime at $0.45/hr. Async pre-recorded is $0.0035. Diarization and other add-ons bill on top."
            }
          }
        },
        {
          "provider": "Modulate",
          "model": "Velma 2",
          "id": "modulate:velma-2-stt-streaming-english-v2",
          "measured": "2026-08-08",
          "status": {
            "label": "GOOD",
            "tone": "good"
          },
          "flag": "Batch measured vendor-direct; streaming re-scored through the gateway.",
          "rec": [
            {
              "label": "Streaming"
            }
          ],
          "cells": {
            "werConv": {
              "v": "12.3%",
              "s": 12.3,
              "tone": "good"
            },
            "wer": {
              "v": "4.4%",
              "s": 4.4,
              "tone": "good",
              "ci": "95% CI 2.8–6.2 · n=50"
            },
            "werStream": {
              "v": "5.4%",
              "s": 5.4,
              "tone": "good",
              "ci": "95% CI 3.1–8.6 · n=50"
            },
            "endpoint": {
              "v": "1.11s",
              "s": 1112,
              "lo": 1112,
              "hi": 1417,
              "ci": "p50–p90 · us-east4 · n=30 · vendor-direct",
              "tone": "bad"
            },
            "ttft": {
              "v": "1.48s",
              "s": 1479,
              "lo": 1479,
              "hi": 1497,
              "ci": "p50–p90 · us-east4 · n=30 · vendor-direct",
              "tone": "warn"
            },
            "cost": {
              "v": "$0.001",
              "s": 0.001,
              "note": "Published streaming rate, $0.06/hr. Batch is $0.03/hr. Not a measurement."
            }
          }
        },
        {
          "provider": "Soniox",
          "model": "stt-rt-v5",
          "id": "soniox:stt-rt-v5",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "rec": [
            {
              "label": "Streaming"
            },
            {
              "label": "Multilingual"
            }
          ],
          "cells": {
            "werConv": {
              "v": "13.1%",
              "s": 13.1,
              "tone": "good",
              "note": "Was 36.5% until a teardown bug of ours stopped finalizing unheard audio."
            },
            "wer": {
              "v": "7.5%",
              "s": 7.5,
              "tone": "warn",
              "ci": "95% CI 4.8–10.7 · n=50"
            },
            "werStream": {
              "v": "7.3%",
              "s": 7.3,
              "tone": "warn"
            },
            "endpoint": {
              "v": "78ms",
              "s": 78,
              "lo": 78,
              "hi": 96,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "good"
            },
            "ttft": {
              "v": "1.04s",
              "s": 1039,
              "lo": 1039,
              "hi": 1679,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "$0.002",
              "s": 0.002
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "GPT-4o Transcribe",
          "id": "openai:gpt-4o-transcribe",
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "rec": [
            {
              "label": "Accuracy"
            },
            {
              "label": "Healthcare",
              "caution": true
            }
          ],
          "cells": {
            "werConv": {
              "v": "31.9%",
              "s": 31.9,
              "tone": "bad",
              "note": "Emits roughly one sentence and stops: 24.0% of its 31.9% is pure deletion against only 6.7% substitution, so it hears fine and quits early. Confirmed against the vendor API directly, none of our code in the path."
            },
            "wer": {
              "v": "2.3%",
              "s": 2.3,
              "tone": "good"
            },
            "werStream": {
              "v": "5.8%",
              "s": 5.8,
              "tone": "warn"
            },
            "endpoint": {
              "v": "572ms",
              "s": 572,
              "lo": 572,
              "hi": 800,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "bad"
            },
            "ttft": {
              "v": "5.06s",
              "s": 5059,
              "lo": 5059,
              "hi": 10967,
              "ci": "p50–p90 · us-east4",
              "tone": "bad"
            },
            "cost": {
              "v": "$0.006",
              "s": 0.006
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "GPT Transcribe",
          "id": "openai:gpt-transcribe",
          "slug": "openai-gpt-transcribe",
          "measured": "2026-07-03",
          "rec": [],
          "cells": {
            "werConv": {
              "v": "15.1%",
              "s": 15.1,
              "tone": "warn",
              "ci": "95% CI 9.7–19.9 · n=99",
              "note": "Deletion-heavy: 6.7% deletions against 7.7% substitutions, at 94.5% coverage — it drops words rather than mishearing them. Hallucinated a tail on one clip (edacc-0070, “…we can do it. Hello, hello.”). Best on Indian English at 11.0% and worst on Southern British at 23.9%."
            },
            "wer": {
              "v": "2.5%",
              "s": 2.5,
              "tone": "good"
            },
            "werStream": {
              "v": "-",
              "s": null
            },
            "endpoint": {
              "v": "-",
              "s": null
            },
            "ttft": {
              "v": "-",
              "s": null
            },
            "cost": {
              "v": "$0.0045",
              "s": 0.0045,
              "note": "Billed per minute, not per token."
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "GPT Live Transcribe",
          "id": "openai:gpt-live-transcribe",
          "slug": "openai-gpt-transcribe-gpt-live-transcribe",
          "measured": "2026-07-21",
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "rec": [
            {
              "label": "Streaming"
            },
            {
              "label": "Real-time phone"
            }
          ],
          "cells": {
            "werConv": {
              "v": "11.3%",
              "s": 11.3,
              "tone": "good"
            },
            "wer": {
              "v": "-",
              "s": null
            },
            "werStream": {
              "v": "4.5%",
              "s": 4.5,
              "tone": "good"
            },
            "endpoint": {
              "v": "1124ms",
              "s": 1124,
              "lo": 1124,
              "hi": 1946,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "bad"
            },
            "ttft": {
              "v": "1.77s",
              "s": 1769,
              "lo": 1769,
              "hi": 2076,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "$0.017",
              "s": 0.017,
              "tone": "bad",
              "note": "gpt-live-transcribe, the streaming path this row is recommended for."
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "GPT-4o-mini Transcribe",
          "id": "openai:gpt-4o-mini-transcribe",
          "status": {
            "label": "EXCELLENT",
            "tone": "good"
          },
          "rec": [
            {
              "label": "Value"
            }
          ],
          "cells": {
            "werConv": {
              "v": "18.5%",
              "s": 18.5,
              "tone": "warn"
            },
            "wer": {
              "v": "2.7%",
              "s": 2.7,
              "tone": "good"
            },
            "werStream": {
              "v": "6.4%",
              "s": 6.4,
              "tone": "warn"
            },
            "endpoint": {
              "v": "460ms",
              "s": 460,
              "lo": 460,
              "hi": 585,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "warn"
            },
            "ttft": {
              "v": "4.98s",
              "s": 4978,
              "lo": 4978,
              "hi": 11206,
              "ci": "p50–p90 · us-east4",
              "tone": "bad"
            },
            "cost": {
              "v": "$0.003",
              "s": 0.003
            }
          }
        },
        {
          "provider": "xAI",
          "model": "Grok STT",
          "id": "xai:stt",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "werConv": {
              "v": "11.3%",
              "s": 11.3,
              "tone": "good",
              "note": "Was 32.9% burst-fed; its live endpointer needs real-time pacing."
            },
            "wer": {
              "v": "4.8%",
              "s": 4.8,
              "tone": "warn",
              "note": "Loudness-normalized best-case; degrades on raw input."
            },
            "werStream": {
              "v": "10.9%",
              "s": 10.9,
              "tone": "warn",
              "note": "Raw audio, n=31: 15/50 very-quiet FLEURS clips excluded where the model returns no transcript (loudness-sensitive; batch 4.8% is loudness-normalized)."
            },
            "endpoint": {
              "v": "305ms",
              "s": 305,
              "lo": 305,
              "hi": 356,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "warn"
            },
            "ttft": {
              "v": "1.34s",
              "s": 1341,
              "lo": 1341,
              "hi": 1392,
              "ci": "p50–p90 · us-east4",
              "tone": "good"
            },
            "cost": {
              "v": "$0.0033",
              "s": 0.0033,
              "note": "grok-stt streaming at $0.20/hr. REST is $0.0017."
            }
          }
        },
        {
          "provider": "Gradium",
          "model": "Gradium ASR",
          "id": "gradium:default",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "werConv": {
              "v": "22.9%",
              "s": 22.9,
              "tone": "bad",
              "note": "Drops far more than it mishears: 13.5% deletions against 8.4% substitutions."
            },
            "wer": {
              "v": "8.4%",
              "s": 8.4,
              "tone": "warn"
            },
            "werStream": {
              "v": "11.7%",
              "s": 11.7,
              "tone": "warn"
            },
            "endpoint": {
              "v": "334ms",
              "s": 334,
              "lo": 334,
              "hi": 406,
              "ci": "p50–p90 · us-east4 · n=30",
              "tone": "warn"
            },
            "ttft": {
              "v": "1.83s",
              "s": 1830,
              "lo": 1830,
              "hi": 2716,
              "ci": "p50–p90 · us-east4",
              "tone": "warn"
            },
            "cost": {
              "v": "$0.0104",
              "s": 0.0104,
              "note": "3 credits/sec on the XS tier, the cheapest paid plan; $0.0068 at the L tier."
            }
          }
        },
        {
          "provider": "Gladia",
          "model": "Solaria-1",
          "id": "gladia:solaria-1",
          "measured": "2026-08-07",
          "meta": "Gladia’s higher-accuracy Solaria-3 has no live endpoint — pre-recorded only.",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "flag": "Probed vendor-direct — Gladia is wired to the gateway but not key-provisioned there yet.",
          "rec": [],
          "cells": {
            "werConv": {
              "v": "18.2%",
              "s": 18.2,
              "tone": "warn"
            },
            "wer": {
              "v": "5.0%",
              "s": 5,
              "tone": "warn"
            },
            "werStream": {
              "v": "11.4%",
              "s": 11.4,
              "tone": "warn",
              "note": "Gladia finalizes per utterance at a 0.05s silence default, so one sentence arrives as several finals; on 5 of 50 clips a middle segment never arrives at all, which is most of the gap to its 5.0% batch score."
            },
            "endpoint": {
              "v": "596ms",
              "s": 596,
              "lo": 596,
              "hi": 1001,
              "ci": "p50–p90 · us-east4 · n=30 · vendor-direct",
              "tone": "bad",
              "note": "Gladia’s own endpointing at the vendor default (0.05s of silence), untuned. 4 of 30 clips finalized early."
            },
            "ttft": {
              "v": "1.49s",
              "s": 1494,
              "lo": 1494,
              "hi": 2099,
              "ci": "p50–p90 · us-east4 · vendor-direct",
              "tone": "good",
              "note": "Partials are off by default upstream; enabled explicitly, without which this would be time-to-first-final, not first token."
            },
            "cost": {
              "v": "$0.0125",
              "s": 0.0125,
              "note": "Starter pay-as-you-go real-time ($0.75/hr). Pre-recorded is $0.0102/min, and committed Growth pricing goes to $0.0042."
            }
          }
        },
        {
          "provider": "Meta",
          "model": "Muse Voice Transcribe",
          "id": "meta:muse-voice-transcribe-1.0",
          "measured": "2026-09-02",
          "status": {
            "label": "GOOD",
            "tone": "good"
          },
          "flag": "First token is measured on the pinned us-east4 vantage. Finalize is measured but withheld -- the speech-end marker in this corpus is less precise than the finalize it is trying to time, so the cell would report marker error.",
          "rec": [
            {
              "label": "Streaming"
            },
            {
              "label": "Value"
            }
          ],
          "cells": {
            "werConv": {
              "v": "11.7%",
              "s": 11.7,
              "tone": "good",
              "ci": "95% CI 8.1–14.6 · n=99",
              "note": "Lowest substitution rate in this run at 4.9%, under the leader’s 5.6% -- it hears better and loses on 5.7% deletions. Scores 8.7% on Indian English against 13.7% on Mainstream US: a 5.0pp inversion, the widest in the run, where the next is 3.6pp and three arms do not invert at all. Numeral rendering is not deterministic dial to dial, and it reformats dates -- \"on 15 august 1940\" came back as \"On August 15, 1940\" -- so formatting variance rides in any WER on this model."
            },
            "wer": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not measured. Meta ships a prerecorded endpoint; no batch sweep has been run on it."
            },
            "werStream": {
              "v": "4.6%",
              "s": 4.63,
              "tone": "good",
              "ci": "95% CI 2.9–6.6 · n=50",
              "note": "Replaces a withheld n=20 reading of 5.3%. Numeral rendering is not deterministic dial to dial and dates are reformatted, so formatting variance rides in any WER on this model."
            },
            "endpoint": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Measured but withheld: on 9 of 30 clips the speech-end marker lands up to 2.4s after Meta committed the turn, while the transcript is complete -- so the marker, not the model, sets the number. Muse finalizes faster than this corpus can resolve."
            },
            "ttft": {
              "v": "1.64s",
              "s": 1643,
              "lo": 1643,
              "hi": 2054,
              "ci": "p50–p90 · us-east4 · n=30 · vendor-direct",
              "tone": "good"
            },
            "cost": {
              "v": "$0.003",
              "s": 0.003,
              "note": "Vendor list price: $3.00 per 1,000 minutes, equivalently $0.18/hr. Streaming and batch are priced identically -- the table does not differentiate -- and there is a single tier, no plan or volume rates."
            }
          }
        },
        {
          "provider": "Google",
          "model": "Gemini 3.5 Transcribe Live",
          "id": "gemini:gemini-3-5-transcribe-live",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "flag": "Vendor-direct only -- not wired to the Speko gateway, so this row measures the model and not the Speko product. Finalize and first token are measured on the pinned us-east4 vantage (n=30).",
          "rec": [
            {
              "label": "Streaming"
            }
          ],
          "cells": {
            "werConv": {
              "v": "13.1%",
              "s": 13.1,
              "tone": "good",
              "note": "The batch sibling `gemini-3.5-transcribe` is listed in the Gemini API and returns ZERO output tokens on every request shape tried, while gemini-3.5-flash transcribes the same bytes in the same body — so only the Live model is measurable. 4.9% substitutions puts its hearing third on the board; it loses on 6.9% deletions. Announced 2026-08-26."
            },
            "endpoint": {
              "v": "1.22s",
              "s": 1218,
              "lo": 1218,
              "hi": 1479,
              "ci": "p50–p90 · us-east4 · n=30 · vendor-direct",
              "tone": "bad"
            },
            "ttft": {
              "v": "1.88s",
              "s": 1882,
              "lo": 1882,
              "hi": 2228,
              "ci": "p50–p90 · us-east4 · n=30 · vendor-direct",
              "tone": "warn"
            },
            "cost": {
              "v": "~$0.009",
              "s": 0.009,
              "note": "ESTIMATE, not a published per-minute rate. Google prices this model per token -- $3.50/1M audio in, $21.00/1M text out -- and the ~$0.005/min + ~$0.004/min figures on the pricing page are Google's own estimate at 25 audio tokens/sec input and 175 text tokens/min output, for a blended ~$0.009/min. Speech density moves it, so read it as an order of magnitude rather than a rate card. Every other price on this column is the vendor's published per-minute rate."
            }
          }
        }
      ],
      "reading": ""
    },
    {
      "id": "stt-codeswitch",
      "navLabel": "Code-switching",
      "identityLabel": "Provider / Model",
      "title": "Code-switching",
      "subtitle": "Two languages in one utterance — error rate and whether the second language survives, per pair, over the live stream.",
      "live": true,
      "meta": "",
      "rankCol": "pairs",
      "headline": [
        "pairs",
        "zh",
        "es"
      ],
      "columns": [
        {
          "id": "pairs",
          "label": "Pairs handled",
          "dir": "higher",
          "info": "Language pairs where the second language survives (both-language coverage over 50%), of 4 tested."
        },
        {
          "id": "zh",
          "label": "EN ↔ ZH",
          "dir": "lower",
          "info": "Mixed-token error rate on Mandarin-English code-switch (ASCEND, streaming); caption = both-language coverage. Lower is better."
        },
        {
          "id": "es",
          "label": "EN ↔ ES",
          "dir": "lower",
          "info": "Word error rate on English-Spanish code-switch (FLEURS concat, streaming); caption = coverage. Lower is better."
        },
        {
          "id": "de",
          "label": "EN ↔ DE",
          "dir": "lower",
          "info": "Word error rate on English-German code-switch (FLEURS concat, streaming); caption = coverage. Lower is better."
        },
        {
          "id": "fr",
          "label": "EN ↔ FR",
          "dir": "lower",
          "info": "Word error rate on English-French code-switch (FLEURS concat, streaming); caption = coverage. Lower is better."
        }
      ],
      "rows": [
        {
          "provider": "Soniox",
          "model": "stt-rt-v5",
          "id": "soniox:stt-rt-v5",
          "measured": "2026-07-21",
          "flag": "",
          "rec": [],
          "cells": {
            "pairs": {
              "v": "4 / 4",
              "s": 4,
              "tone": "good"
            },
            "zh": {
              "v": "11.8%",
              "s": 11.8,
              "tone": "good",
              "ci": "88% both-lang"
            },
            "es": {
              "v": "4.9%",
              "s": 4.9,
              "tone": "good",
              "ci": "97% both-lang"
            },
            "de": {
              "v": "5.1%",
              "s": 5.1,
              "tone": "good",
              "ci": "100% both-lang"
            },
            "fr": {
              "v": "8.8%",
              "s": 8.8,
              "tone": "good",
              "ci": "89% both-lang"
            }
          }
        },
        {
          "provider": "AssemblyAI",
          "model": "universal-3-5-pro",
          "id": "assemblyai:universal-3-5-pro",
          "measured": "2026-07-31",
          "flag": "",
          "rec": [],
          "cells": {
            "pairs": {
              "v": "4 / 4",
              "s": 4,
              "tone": "good"
            },
            "zh": {
              "v": "9.8%",
              "s": 9.8,
              "tone": "good",
              "ci": "100% both-lang"
            },
            "es": {
              "v": "3.5%",
              "s": 3.5,
              "tone": "good",
              "ci": "100% both-lang"
            },
            "de": {
              "v": "3.7%",
              "s": 3.7,
              "tone": "good",
              "ci": "98% both-lang"
            },
            "fr": {
              "v": "2.6%",
              "s": 2.6,
              "tone": "good",
              "ci": "93% both-lang"
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "scribe_v2_realtime",
          "id": "elevenlabs:scribe_v2_realtime",
          "measured": "2026-07-21",
          "flag": "Strong on Latin-script pairs; anglicizes Mandarin (drops it to English).",
          "rec": [],
          "cells": {
            "pairs": {
              "v": "3 / 4",
              "s": 3,
              "tone": "warn"
            },
            "zh": {
              "v": "75.0%",
              "s": 75,
              "tone": "bad",
              "ci": "34% both-lang"
            },
            "es": {
              "v": "3.8%",
              "s": 3.8,
              "tone": "good",
              "ci": "100% both-lang"
            },
            "de": {
              "v": "3.1%",
              "s": 3.1,
              "tone": "good",
              "ci": "98% both-lang"
            },
            "fr": {
              "v": "7.8%",
              "s": 7.8,
              "tone": "good",
              "ci": "89% both-lang"
            }
          }
        }
      ],
      "reading": ""
    },
    {
      "id": "tts",
      "navLabel": "Text-to-Speech",
      "identityLabel": "System / Model",
      "title": "Text-to-Speech",
      "subtitle": "Naturalness, drift, latency, robustness and cost for every text-to-speech model we measure.",
      "live": true,
      "meta": "",
      "measured": "2026-07-03",
      "headline": [
        "natural",
        "synth",
        "cost"
      ],
      "scatter": {
        "x": "synth",
        "y": "natural",
        "cornerLabel": "fast + natural",
        "note": "naturalness = arena Elo from our own blind human A/B study on phone-agent lines; field mean 1500"
      },
      "columns": [
        {
          "id": "natural",
          "label": "Naturalness",
          "dir": "higher",
          "info": "Arena Elo from blind human A/B votes on phone-agent lines; field mean 1500, higher = preferred."
        },
        {
          "id": "ci",
          "label": "Win % · 95% CI",
          "dir": "higher",
          "info": "Head-to-head win rate against the rest of the field, with its 95% confidence interval. Overlapping intervals = statistical tie — read the interval, not just the point. Higher = preferred."
        },
        {
          "id": "drift",
          "label": "Drift",
          "dir": "lower",
          "info": "Voice steadiness: mean pairwise embedding distance over 20 repeats. Lower = steadier. \"det\" = deterministic output. \"—\" = not yet run for this model."
        },
        {
          "id": "synth",
          "label": "Synth p50",
          "dir": "lower",
          "info": "Time to first audio, p50 (n=30, us-east4, sequential). Lower is better."
        },
        {
          "id": "robustness",
          "label": "Robustness",
          "dir": "higher",
          "info": "Does it say the right WORDS for a number, date, currency amount or operator — a different question from drift, which is whether the voice sounds like itself. Scored 0–1 by fining each defect on its severity (fatal / serious / minor) and on how often a caller hits that input class. Higher = fewer defects. \"us-only\" / \"wide-us\" rows ran fewer fixtures and are optimistic against full-basis rows, so a small gap across bases is not meaningful."
        },
        {
          "id": "cost",
          "label": "Cost / 1M chars",
          "dir": "lower",
          "info": "$ per 1M characters, vendor list. Pay-as-you-go where offered, otherwise the cheapest paid plan. Lower = cheaper; \"~\" = converted from token or hourly pricing at 900 chars/min of speech."
        }
      ],
      "rows": [
        {
          "provider": "ElevenLabs",
          "model": "eleven_v3_conversational",
          "id": "elevenlabs:eleven_v3_conversational",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1590",
              "s": 1590,
              "tone": "good"
            },
            "ci": {
              "v": "66%",
              "s": 66,
              "tone": "good",
              "ci": "60–72%"
            },
            "drift": {
              "v": "20",
              "s": 20
            },
            "synth": {
              "v": "266ms",
              "s": 266,
              "lo": 265,
              "hi": 268,
              "ci": "265–268 two-sweep p50"
            },
            "cost": {
              "v": "$100",
              "s": 100,
              "tone": "warn"
            },
            "robustness": {
              "v": "0.24",
              "s": 24,
              "tone": "bad",
              "tie": 3,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of 99 cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: 45,090 outstanding.”\n• minor · dates — kept the value but never said “july”; a prompt fixes this (7 of 23 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment is on seven fifteen, twenty twenty six.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (6 of 23 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten 'o two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is o three zero nine this year.”\n• minor · dates — kept the value but never said “february”; a prompt fixes this (9 of 23 transcripts)\n→ we sent “Renewal is 02/29/2024.”\n→ it said “Renewal is two twenty nine twenty twenty four.”\n• serious · dates (international, ×0.4) — never said “march” (4 of 7 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on twenty twenty six nine.”\n• minor · dates (international, ×0.4) — kept the value but never said “june”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends twenty twenty five-six thirty.”\n• minor · math operators — kept the value but never said “minus”; a prompt fixes this (6 of 7 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten-four equals six remaining.”\n• fatal · times — never said “a m” (3 of 7 transcripts)\n→ we sent “We open at 9:00am.”\n→ it said “We open at nine.”"
            }
          }
        },
        {
          "provider": "Gemini",
          "model": "gemini-3.1-flash-tts-preview",
          "id": "google-tts:gemini-3.1-flash-tts-preview",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1591",
              "s": 1591,
              "tone": "good"
            },
            "ci": {
              "v": "66%",
              "s": 66,
              "tone": "good",
              "ci": "59–72%"
            },
            "drift": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "synth": {
              "v": "978ms",
              "s": 978,
              "lo": 978,
              "hi": 1235,
              "ci": "978–1235 p50–p90"
            },
            "cost": {
              "v": "~$33.3",
              "s": 33.3,
              "note": "Token-billed: $1/1M text in, $20/1M audio out; Google states 25 audio tokens/sec (ai.google.dev/gemini-api/docs/pricing). At 900 chars/min of speech that is ~$33 per 1M chars; text-in adds ~$0.25, negligible."
            },
            "robustness": {
              "v": "0.68",
              "s": 68,
              "tone": "warn",
              "tie": 3,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of $0.99 was applied.”\n• minor · dates — kept the value but never said “july”; a prompt fixes this (7 of 23 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment is on seven fifteen, twenty twenty six.”\n• minor · dates — kept the value but never said “january”; a prompt fixes this (8 of 23 transcripts)\n→ we sent “Renewal falls on 1/1/2027.”\n→ it said “Renewal falls on one one twenty twenty seven.”\n• minor · dates — kept the value but never said “november”; a prompt fixes this (7 of 23 transcripts)\n→ we sent “The policy started 11/30/2024.”\n→ it said “The policy started eleven thirty, twenty twenty four.”\n• minor · dates (international, ×0.4) — kept the value but never said “march”; a prompt fixes this (10 of 23 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on twenty twenty six zero three zero nine.”"
            }
          }
        },
        {
          "provider": "Deepgram",
          "model": "aura-2",
          "id": "deepgram:aura-2",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1584",
              "s": 1584,
              "tone": "good"
            },
            "ci": {
              "v": "65%",
              "s": 65,
              "tone": "good",
              "ci": "58–70%"
            },
            "drift": {
              "v": "9",
              "s": 9,
              "tone": "good"
            },
            "synth": {
              "v": "125ms",
              "s": 125,
              "lo": 125,
              "hi": 138,
              "ci": "125–138 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$30.0",
              "s": 30,
              "tone": "good"
            },
            "robustness": {
              "v": "0.41",
              "s": 41,
              "tone": "bad",
              "tie": 2,
              "note": "Fined:\n• minor · currency (international, ×0.4) — kept the value but never said “forty five”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Negative four thousand five hundred ninety euros outstanding.”\n• minor · dates — kept the value but never said “july”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment is on zero seven fifteen, twenty twenty six.”\n• minor · dates — kept the value but never said “january”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Renewal falls on 1/1/2027.”\n→ it said “Renewal falls on one one twenty twenty seven.”\n• minor · dates — kept the value but never said “november”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The policy started 11/30/2024.”\n→ it said “The policy started eleven thirty twenty twenty four.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten 'o two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• minor · dates (international, ×0.4) — kept the value but never said “march”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on twenty twenty six o three zero nine.”\n• minor · dates (international, ×0.4) — kept the value but never said “june”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends twenty twenty five o six thirty.”\n• fatal · large numbers — never said “zero zero” (5 of 5 transcripts)\n→ we sent “Only 007 seats remain.”\n→ it said “Only seven seats remain.”\n• minor · math operators — kept the value but never said “minus”; a prompt fixes this (3 of 5 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten-four equals six remaining.”\n• minor · math operators — kept the value but never said “three quarters”; a prompt fixes this (5 of 5 transcripts)\n→ we sent “Give me 3/4 of the batch.”\n→ it said “Give me three four of the batch.”\n• minor · math operators — kept the value but never said “times”; a prompt fixes this (3 of 5 transcripts)\n→ we sent “Check it: 7×8=56, not 54.”\n→ it said “Check it, seven x eight equals fifty-six, not fifty-four.”\n• fatal · math operators — never said “divided” (4 of 5 transcripts)\n→ we sent “That is 100÷4=25 per person.”\n→ it said “That is 104 equals 25 per person.”\n• minor · math operators — kept the value but never said “two thirds”; a prompt fixes this (5 of 5 transcripts)\n→ we sent “Use 2/3 cup of sugar.”\n→ it said “Use two-three cup of sugar.”"
            }
          }
        },
        {
          "provider": "Deepgram",
          "model": "flux",
          "id": "deepgram:flux",
          "rec": [],
          "cells": {
            "natural": {
              "v": "~1550",
              "s": 1550
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "25",
              "s": 25,
              "tone": "warn",
              "note": "variance analyzer, 20 repeats, run 2026-09-02T10-00-22-058Z, measured 2026-09-02. Single draw; not comparable with other rows in this column."
            },
            "synth": {
              "v": "106ms",
              "s": 106,
              "lo": 106,
              "hi": 148,
              "ci": "106–148 p50–p90",
              "tone": "good",
              "note": "Mean of two 2026-08-12 sweeps (104/108 p50), n=30 each, concurrency 1, one keep-alive connection, 0 rejected. Anchors in the same sweeps came in 11–26% off their published cells, so the column needs a full re-baseline before these are read against each other."
            },
            "cost": {
              "v": "$45.0",
              "s": 45,
              "tone": "warn",
              "note": "Deepgram published list, Pay As You Go: $0.0450 per 1,000 characters = $45/1M. Same unit that puts Aura-2 at $0.030/1k = the $30 cell above, so the two are directly comparable — Flux TTS costs 50% more than Aura-2. Growth tier is $0.0405/1k ($40.5/1M). Free to build against through 2026-09-12 (45 concurrent streams globally, 5 EU/AU); billing starts 09-13."
            },
            "robustness": {
              "v": "0.41",
              "s": 41,
              "tone": "bad",
              "tie": 1,
              "note": "Fined:\n• minor · currency (international, ×0.4) — kept the value but never said “forty five”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: negative four thousand five hundred and ninety euros outstanding.”\n• minor · dates — kept the value but never said “july”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment is on zero seven fifteen, twenty twenty six.”\n• minor · dates — kept the value but never said “january”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Renewal falls on 1/1/2027.”\n→ it said “Renewal falls on one one twenty twenty seven.”\n• minor · dates — kept the value but never said “november”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The policy started 11/30/2024.”\n→ it said “The policy started eleven thirty, twenty twenty four.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten zero two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (3 of 5 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• minor · dates — kept the value but never said “february”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Renewal is 02/29/2024.”\n→ it said “Renewal is zero two twenty nine twenty twenty four.”\n• minor · dates (international, ×0.4) — kept the value but never said “march”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on twenty twenty six zero three zero nine.”\n• minor · dates (international, ×0.4) — kept the value but never said “june”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends twenty twenty five zero six thirty.”\n• fatal · large numbers — never said “zero zero” (5 of 5 transcripts)\n→ we sent “Only 007 seats remain.”\n→ it said “Only seven seats remain.”\n• minor · math operators — kept the value but never said “minus”; a prompt fixes this (3 of 5 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten, four equals six remaining.”\n• minor · math operators — kept the value but never said “three quarters”; a prompt fixes this (5 of 5 transcripts)\n→ we sent “Give me 3/4 of the batch.”\n→ it said “Give me three-four of the batch.”\n• fatal · math operators — never said “times” (3 of 5 transcripts)\n→ we sent “Check it: 7×8=56, not 54.”\n→ it said “Check it. 7 psi equals 56, not 54.”\n• fatal · math operators — never said “divided” (5 of 5 transcripts)\n→ we sent “That is 100÷4=25 per person.”\n→ it said “That is 104 equals 25 per person.”"
            }
          }
        },
        {
          "provider": "Cartesia",
          "model": "sonic-3.5",
          "id": "cartesia:sonic-3.5",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1574",
              "s": 1574,
              "tone": "good"
            },
            "ci": {
              "v": "65%",
              "s": 65,
              "tone": "good",
              "ci": "58–71%"
            },
            "drift": {
              "v": "13",
              "s": 13,
              "tone": "good"
            },
            "synth": {
              "v": "121ms",
              "s": 121,
              "lo": 121,
              "hi": 236,
              "ci": "121–236 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "note": "1 credit/char on Pro, the cheapest paid plan; $39.2 at Startup, $37.4 at Scale. Cloned voices bill 1.5x."
            },
            "robustness": {
              "v": "0.55",
              "s": 55.00000000000001,
              "tone": "warn",
              "tie": 2,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is $12.50.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of 99 cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: €45,90 outstanding.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (6 of 23 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten zero two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• minor · units — kept the value but never said “two hundred fifty” or “gigabyte”; a prompt fixes this (6 of 7 transcripts)\n→ we sent “You have 250GB remaining.”\n→ it said “You have two five zero GB remaining.”\n• minor · units — said “slash” out loud; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Wind is 25km/h northwest.”\n→ it said “Wind is 25 kilometers slash h northwest.”"
            }
          }
        },
        {
          "provider": "Cartesia",
          "model": "sonic-3.6",
          "id": "cartesia:sonic-3.6",
          "rec": [],
          "measured": "2026-08-28",
          "flag": "Naturalness is an auto-MOS estimate, not a listening-panel result. Synth is vendor-direct US East.",
          "cells": {
            "natural": {
              "v": "~1578",
              "s": 1578,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "10",
              "s": 10,
              "tone": "good"
            },
            "synth": {
              "v": "120ms",
              "s": 120,
              "lo": 120,
              "hi": 135,
              "ci": "120–135 p50–p90 · US East · vendor-direct · n=400",
              "tone": "good",
              "note": "Vendor-direct from us-east4, n=400 (2 sweeps x 200), concurrency 1, one keep-alive socket."
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "note": "Derived, not published: Cartesia documents \"approximately 1 credit per character\" and Pro at $5/100K credits. No per-character USD rate exists. Cloned voices bill 1.5x."
            },
            "robustness": {
              "v": "0.50",
              "s": 50,
              "tone": "warn",
              "tie": 1,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is twelve dollars and fifty cents.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of 99 cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: €45.90 outstanding.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (7 of 23 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten zero two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• fatal · dates (international, ×0.4) — never said “june” (7 of 7 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends 202 May 6, 2030.”"
            }
          }
        },
        {
          "provider": "Soniox",
          "model": "tts-rt-v1",
          "id": "soniox:tts-rt-v1",
          "routable": false,
          "rec": [],
          "cells": {
            "natural": {
              "v": "1569",
              "s": 1569,
              "tone": "good"
            },
            "ci": {
              "v": "63%",
              "s": 63,
              "tone": "good",
              "ci": "56–69%"
            },
            "drift": {
              "v": "24",
              "s": 24,
              "tone": "warn"
            },
            "synth": {
              "v": "362ms",
              "s": 362,
              "lo": 362,
              "hi": 388,
              "ci": "362–388 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "~$13.0",
              "s": 13,
              "tone": "good",
              "note": "Token-billed: $4/1M text in, $21.50/1M audio out; the vendor states ~$0.70 per hour of generated speech (soniox.com/pricing). At 900 chars/min (54,000 chars/hour) that is ~$13 per 1M chars."
            },
            "robustness": {
              "v": "0.49",
              "s": 49,
              "tone": "warn",
              "tie": 1,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (4 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of $9.99 was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (1 of 5 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: €45.90 outstanding.”\n• minor · dates — said “slash” out loud; a prompt fixes this (6 of 11 transcripts)\n→ we sent “Renewal falls on 1/1/2027.”\n→ it said “Renewal falls on one slash one slash twenty twenty seven.”\n• minor · dates — said “slash” out loud; a prompt fixes this (6 of 14 transcripts)\n→ we sent “The policy started 11/30/2024.”\n→ it said “The policy started eleven slash thirty slash twenty twenty four.”\n• minor · dates — said “slash” out loud; a prompt fixes this (8 of 14 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten slash zero two.”\n• minor · dates — said “slash” out loud; a prompt fixes this (5 of 14 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three slash zero nine this year.”\n• minor · dates — said “slash” out loud; a prompt fixes this (8 of 14 transcripts)\n→ we sent “Renewal is 02/29/2024.”\n→ it said “Renewal is zero two slash 29 slash 2,024.”\n• minor · dates — said “slash”, “zero seven slash” out loud; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment is on zero seven slash fifteen slash twenty twenty six.”\n• serious · dates (international, ×0.4) — said “slash” out loud (5 of 11 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends 2025. Slash 06Slash30.”\n• minor · math operators — kept the value but never said “minus”; a prompt fixes this (6 of 7 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten to four equals six remaining.”"
            }
          }
        },
        {
          "provider": "Soniox",
          "model": "tts-rt-v2",
          "id": "soniox:tts-rt-v2",
          "rec": [],
          "cells": {
            "natural": {
              "v": "~1606",
              "s": 1606,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "18",
              "s": 18
            },
            "synth": {
              "v": "381ms",
              "s": 381,
              "lo": 381,
              "hi": 417,
              "ci": "381–417 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "~$13.0",
              "s": 13,
              "tone": "good",
              "note": "Token-billed: $4/1M text in, $21.50/1M audio out; the vendor states ~$0.70 per hour of generated speech (soniox.com/pricing). At 900 chars/min (54,000 chars/hour) that is ~$13 per 1M chars. Unchanged from v1 — the pricing page now lists only TTS v2 at the same rates."
            },
            "robustness": {
              "v": "0.78",
              "s": 78,
              "tone": "good",
              "tie": 0,
              "note": "Fined:\n• minor · dates — said “slash” out loud; a prompt fixes this (8 of 14 transcripts)\n→ we sent “Renewal falls on 1/1/2027.”\n→ it said “Renewal falls on one slash one slash 27.”\n• serious · dates — said “slash” out loud (8 of 14 transcripts)\n→ we sent “The policy started 11/30/2024.”\n→ it said “The policy started. Eleventhirty slash twenty twenty four.”\n• minor · dates — said “slash” out loud; a prompt fixes this (8 of 14 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten slash zero two.”\n• minor · dates — said “slash” out loud; a prompt fixes this (11 of 30 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three slash zero nine this year.”\n• minor · dates — said “slash” out loud; a prompt fixes this (8 of 14 transcripts)\n→ we sent “Renewal is 02/29/2024.”\n→ it said “Renewal is zero two slash 29 slash 2,024.”\n• minor · dates — said “slash”, “zero seven slash” out loud; a prompt fixes this (11 of 23 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment is on zero seven slash fifteen slash two thousand twenty six.”\n• minor · dates (international, ×0.4) — said “dash” out loud; a prompt fixes this (3 of 11 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on twenty twenty six dash zero three -nine.”\n• minor · dates (international, ×0.4) — said “slash” out loud; a prompt fixes this (6 of 11 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends 2025 slash zero six slash thirty.”\n• minor · math operators — kept the value but never said “minus”; a prompt fixes this (13 of 23 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten to four equals six remaining.”"
            }
          }
        },
        {
          "provider": "Speechify",
          "model": "simba-3.2",
          "id": "speechify:simba-3.2",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1573",
              "s": 1573,
              "tone": "good"
            },
            "ci": {
              "v": "63%",
              "s": 63,
              "tone": "good",
              "ci": "57–69%"
            },
            "drift": {
              "v": "7",
              "s": 7,
              "tone": "good"
            },
            "synth": {
              "v": "345ms",
              "s": 345,
              "lo": 345,
              "hi": 375,
              "ci": "345–375 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$10.0",
              "s": 10,
              "tone": "good",
              "note": "Vendor list (Starter tier); $6–8 per 1M at Pro/Scale volume."
            },
            "robustness": {
              "v": "0.70",
              "s": 70,
              "tone": "good",
              "tie": 0,
              "note": "Fined:\n• fatal · decimals — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Growth slowed to -0.5% this month.”\n→ it said “Growth slowed to 0.5% this month.”\n• minor · math operators — said “equal to” out loud; a prompt fixes this (4 of 8 transcripts)\n→ we sent “Remember that 2+2=4 always.”\n→ it said “Remember that two plus two equal to four always.”"
            }
          }
        },
        {
          "provider": "Inworld",
          "model": "inworld-tts-2",
          "id": "inworld:inworld-tts-2",
          "flag": "Re-benchmarked 2026-09-03: drift and robustness. Naturalness and time-to-first-audio are unchanged.",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1561",
              "s": 1561,
              "tone": "good"
            },
            "ci": {
              "v": "63%",
              "s": 63,
              "tone": "good",
              "ci": "57–69%"
            },
            "drift": {
              "v": "10",
              "s": 10,
              "tone": "good"
            },
            "synth": {
              "v": "116ms",
              "s": 116,
              "lo": 116,
              "hi": 125,
              "ci": "116–125 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$25.0",
              "s": 25,
              "tone": "good",
              "note": "On-demand. Drops to $20 on Creator and $12.50 on Growth."
            },
            "robustness": {
              "v": "0.93",
              "s": 93,
              "tone": "good",
              "tie": 1,
              "note": "Fined:\n• minor · dates — said “slash” out loud; a prompt fixes this (4 of 7 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to 10. Slash zero two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• minor · dates (international, ×0.4) — said “slash” out loud; a prompt fixes this (4 of 7 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends twenty twenty five slash zero six slash thirty.”"
            }
          }
        },
        {
          "provider": "Inworld",
          "model": "inworld-tts-2-flash",
          "id": "inworld:inworld-tts-2-flash",
          "measured": "2026-09-03",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote.",
          "rec": [],
          "cells": {
            "natural": {
              "v": "~1501",
              "s": 1501,
              "dim": true
            },
            "drift": {
              "v": "13",
              "s": 13,
              "tone": "good"
            },
            "synth": {
              "v": "82ms",
              "s": 82,
              "lo": 82,
              "hi": 89,
              "ci": "82–89 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$15.0",
              "s": 15,
              "tone": "good",
              "note": "On-demand."
            },
            "robustness": {
              "v": "0.93",
              "s": 93,
              "tone": "good",
              "tie": 1,
              "note": "Fined:\n• minor · dates — said “slash” out loud; a prompt fixes this (4 of 7 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to 10 slash zero two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• minor · dates (international, ×0.4) — said “slash” out loud; a prompt fixes this (4 of 7 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends 2025 slash six slash thirty.”"
            }
          }
        },
        {
          "provider": "Fish Audio",
          "model": "s2.1-pro",
          "id": "fishaudio:s2.1-pro",
          "rec": [],
          "cells": {
            "natural": {
              "v": "~1566",
              "s": 1566,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "synth": {
              "v": "185ms",
              "s": 185,
              "lo": 185,
              "hi": 235,
              "ci": "185–235 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$15.0",
              "s": 15,
              "tone": "good",
              "note": "Billed per UTF-8 byte, so this holds for Latin text only — CJK runs ~3x. The s2.1-pro-free tier is $0 through 2026-08-31."
            },
            "robustness": {
              "v": "0.00",
              "s": 0,
              "tone": "bad",
              "tie": 2,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is $12.50.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of 99 cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: forty-five euros ninety outstanding.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten-zero-two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• fatal · decimals — never said “plus or minus” or “zero point zero five” (5 of 7 transcripts)\n→ we sent “Tolerance is ±0.05 mm on that part.”\n→ it said “Tolerance is smaanemems yamunt, zero five millimeter, on that part.”\n• fatal · large numbers — never said “one thousand forty two” (7 of 7 transcripts)\n→ we sent “Your queue position is 1,042.”\n→ it said “Your cue position is Unvergil zero cat do.”\n• fatal · math operators — never said “three to one” (7 of 7 transcripts)\n→ we sent “Mix it at a 3:1 ratio.”\n→ it said “Mix it at a trois-un ratio.”\n• minor · units — kept the value but never said “gigabyte”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “You have 250GB remaining.”\n→ it said “You have two hundred and fifty GB remaining.”"
            }
          }
        },
        {
          "provider": "Smallest",
          "model": "lightning_v3.1",
          "id": "smallest:lightning_v3.1",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1544",
              "s": 1544,
              "tone": "good"
            },
            "ci": {
              "v": "58%",
              "s": 58,
              "tone": "good",
              "ci": "51–64%"
            },
            "drift": {
              "v": "det",
              "s": 0.27,
              "tone": "good"
            },
            "synth": {
              "v": "173ms",
              "s": 173,
              "lo": 173,
              "hi": 193,
              "ci": "173–193 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$25.0",
              "s": 25,
              "tone": "good",
              "note": "Pay-as-you-go (Pro tier); ~$0.25/10k chars."
            },
            "robustness": {
              "v": "0.25",
              "s": 25,
              "tone": "bad",
              "tie": 3,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is $12.50.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of 99 cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: 45 euros and 90 cents outstanding.”\n• minor · dates — kept the value but never said “december”; a prompt fixes this (7 of 7 transcripts)\n→ we sent “We close on Dec 24.”\n→ it said “We close on Deck 24.”\n• minor · math operators — kept the value but never said “minus”; a prompt fixes this (6 of 7 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten to four equals six remaining.”\n• fatal · math operators — never said “divided” (6 of 7 transcripts)\n→ we sent “That is 100÷4=25 per person.”\n→ it said “That is $134 equals 25 per person.”\n• minor · times — kept the value but never said “six oh five”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Boarding runs 6:05 to 6:20.”\n→ it said “Boarding runs six five to six twenty.”\n• minor · units — kept the value but never said “two hundred fifty”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “You have 250GB remaining.”\n→ it said “You have two five zero gigabytes remaining.”"
            }
          }
        },
        {
          "provider": "xAI Grok",
          "model": "grok-tts",
          "id": "xai:tts",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1495",
              "s": 1495
            },
            "ci": {
              "v": "48%",
              "s": 48,
              "ci": "42–55%"
            },
            "drift": {
              "v": "15",
              "s": 15
            },
            "synth": {
              "v": "272ms",
              "s": 272,
              "lo": 272,
              "hi": 325,
              "ci": "272–325 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$15.0",
              "s": 15,
              "tone": "good"
            },
            "robustness": {
              "v": "0.43",
              "s": 43,
              "tone": "bad",
              "tie": 2,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• minor · dates — said “slash” out loud; a prompt fixes this (3 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three slash zero nine this year.”\n• minor · dates (international, ×0.4) — kept the value but never said “march”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on twenty twenty six zero three zero nine.”\n• fatal · decimals — dropped the minus sign (1 of 7 transcripts)\n→ we sent “Growth slowed to -0.5% this month.”\n→ it said “Growth slowed to 0.5% this month.”"
            }
          }
        },
        {
          "provider": "Gradium",
          "model": "Gradium TTS",
          "slug": "gradium-default",
          "id": "gradium:default",
          "rec": [],
          "measured": "2026-09-01",
          "flag": "Pinned to voice Brooklyn (D6COLz20Hw7uh3UK), the roster Gradium recommends since 2026-08-31. Naturalness is inverted from the automated scorer, not fitted from votes, so it carries no interval and is excluded from the tie at the top. The panel’s own 1448 measured the model this one replaced.",
          "cells": {
            "natural": {
              "v": "~1585",
              "s": 1585,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "synth": {
              "v": "272ms",
              "s": 272,
              "tone": "good",
              "note": "Re-measured 2026-09-01 on the new default model, voice Brooklyn: two n=30 sweeps at concurrency 1 on one keep-alive connection, p50 274 and 270. No range — the laptop vantage is p50-comparable but its p90 is inflated. The previous 244ms was the predecessor model on the arena voice, which now measures 276 then 347 across the same two sweeps."
            },
            "cost": {
              "v": "$57.8",
              "s": 57.8,
              "note": "1 credit/char on the XS tier, the cheapest paid plan; $35.9 at the L tier."
            },
            "robustness": {
              "v": "0.00",
              "s": 0,
              "tone": "bad",
              "tie": 10,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (5 of 7 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is twelve dollars off fifty.”\n• fatal · currency — dropped the minus sign (5 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (2 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “The refund of my storage media was applied.”\n• minor · dates — kept the value but never said “july”; a prompt fixes this (8 of 23 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment is on oh seven fifteen, twenty twenty six.”\n• minor · dates — kept the value but never said “december”; a prompt fixes this (5 of 7 transcripts)\n→ we sent “We close on Dec 24.”\n→ it said “We close on Deck 24.”\n• serious · dates — never said “august” or “third” (11 of 23 transcripts)\n→ we sent “Please arrive by Aug 3rd.”\n→ it said “Please arrive by Oxford.”\n• minor · dates (international, ×0.4) — kept the value but never said “march”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on twenty twenty six, oh three oh nine.”\n• fatal · decimals — never said “plus or minus” (4 of 7 transcripts)\n→ we sent “Tolerance is ±0.05 mm on that part.”\n→ it said “Talent is buro, point zero five millimeters on that part.”\n• fatal · large numbers — never said “seven” or “zero zero” (4 of 7 transcripts)\n→ we sent “Only 007 seats remain.”\n→ it said “Kommissar ser sannsyn.”\n• minor · math operators — kept the value but never said “equals”; a prompt fixes this (5 of 7 transcripts)\n→ we sent “Remember that 2+2=4 always.”\n→ it said “Remember that two plus two point four always.”\n• fatal · math operators — never said “minus” or “six” (5 of 7 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten for our night teams' sex remaining.”\n• fatal · math operators — never said “divided” or “twenty five” (6 of 7 transcripts)\n→ we sent “That is 100÷4=25 per person.”\n→ it said “That is 108.425 per person.”\n• fatal · times — never said “six oh five” or “six twenty” (3 of 7 transcripts)\n→ we sent “Boarding runs 6:05 to 6:20.”\n→ it said “Yeah. And then check. So”"
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "gpt-4o-mini-tts",
          "id": "openai:gpt-4o-mini-tts",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1424",
              "s": 1424
            },
            "ci": {
              "v": "37%",
              "s": 37,
              "ci": "31–43%"
            },
            "drift": {
              "v": "22",
              "s": 22,
              "tone": "warn"
            },
            "synth": {
              "v": "691ms",
              "s": 691,
              "lo": 691,
              "hi": 1171,
              "ci": "691–1171 p50–p90"
            },
            "cost": {
              "v": "~$20.0",
              "s": 20,
              "tone": "good",
              "note": "Token-billed: $0.60/1M text in, $12/1M audio out; OpenAI publishes no per-character or per-minute rate. At ~25 audio tokens/sec (not OpenAI-published) and 900 chars/min, ~$20 per 1M chars. Legacy tts-1 is $15, tts-1-hd $30 per 1M chars."
            },
            "robustness": {
              "v": "0.93",
              "s": 93,
              "tone": "good",
              "tie": 2,
              "note": "Fined:\n• minor · dates — kept the value but never said “july”; a prompt fixes this (8 of 16 transcripts)\n→ we sent “Your appointment is on 07/15/2026.”\n→ it said “Your appointment. Is on seven fifteen wenty twenty six.”\n• minor · dates — kept the value but never said “january”; a prompt fixes this (3 of 5 transcripts)\n→ we sent “Renewal falls on 1/1/2027.”\n→ it said “Renewal falls on one one. Etwenty twenty seven.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten 'o two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (5 of 5 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your SWAT is zero three zero nine this year.”\n• minor · dates — kept the value but never said “february”; a prompt fixes this (7 of 16 transcripts)\n→ we sent “Renewal is 02/29/2024.”\n→ it said “Renewal is zero two twenty nine twenty twenty four.”\n• minor · dates (international, ×0.4) — said “dash” out loud; a prompt fixes this (3 of 5 transcripts)\n→ we sent “The order shipped on 2026-03-09.”\n→ it said “The order shipped on two thousand twenty-six dash zero three dash zero nine.”\n• minor · dates (international, ×0.4) — kept the value but never said “june”; a prompt fixes this (5 of 5 transcripts)\n→ we sent “The lease ends 2025/06/30.”\n→ it said “The lease ends twenty twenty five six thirty.”"
            }
          }
        },
        {
          "provider": "Rime",
          "model": "arcanav3",
          "id": "rime:arcanav3",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1429",
              "s": 1429
            },
            "ci": {
              "v": "36%",
              "s": 36,
              "ci": "31–42%"
            },
            "drift": {
              "v": "21",
              "s": 21
            },
            "synth": {
              "v": "238ms",
              "s": 238,
              "lo": 238,
              "hi": 248,
              "ci": "238–248 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$40.0",
              "s": 40,
              "note": "Arcana on Starter; $30 on Growth. The vendor pricing page contradicts itself — its FAQ quotes $0.05/1k and its card quotes the cheaper Mist rate."
            },
            "robustness": {
              "v": "0.50",
              "s": 50,
              "tone": "warn",
              "tie": 4,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is twelve dollars fifty cents.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of ninety-nine cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: €45,90 outstanding.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (7 of 23 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten zero two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three zero nine this year.”\n• minor · large numbers — kept the value but never said “twelve thousand”; a prompt fixes this (7 of 7 transcripts)\n→ we sent “We shipped 12,003 units.”\n→ it said “We shipped one-two-zero-zero-three units.”\n• minor · large numbers — kept the value but never said “three point seven five million”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “We reviewed 3,750,000 documents.”\n→ it said “We reviewed three seven five zero zero zero zero zero zero documents.”\n• minor · large numbers — kept the value but never said “one million”; a prompt fixes this (9 of 23 transcripts)\n→ we sent “We shipped 1,000,000 units last year.”\n→ it said “We shipped one zero zero zero zero zero zero zero zero units last year.”\n• minor · math operators — said “equal sign”, “equals sign” out loud; a prompt fixes this (6 of 7 transcripts)\n→ we sent “Remember that 2+2=4 always.”\n→ it said “Remember that 2 plus 2 equals sign 4 always.”"
            }
          }
        },
        {
          "provider": "MiniMax",
          "model": "speech-2.8-hd",
          "id": "minimax:speech-2.8-hd",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1431",
              "s": 1431
            },
            "ci": {
              "v": "36%",
              "s": 36,
              "ci": "30–43%"
            },
            "drift": {
              "v": "12",
              "s": 12,
              "tone": "good"
            },
            "synth": {
              "v": "294ms",
              "s": 294,
              "lo": 294,
              "hi": 354,
              "ci": "294–354 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$100.0",
              "s": 100,
              "tone": "warn",
              "note": "HD tier; speech-2.8-turbo ≈ $60/1M."
            },
            "robustness": {
              "v": "0.00",
              "s": 0,
              "tone": "bad",
              "tie": 0,
              "note": "Fined:\n• minor · currency (international, ×0.4) — kept the value but never said “yen”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “The fare is ¥12,800 one way.”\n→ it said “The fare is twelve thousand eight hundred yuan one way.”\n• minor · dates — kept the value but never said “december”; a prompt fixes this (7 of 7 transcripts)\n→ we sent “We close on Dec 24.”\n→ it said “We close on Deck twenty-four.”\n• serious · dates — never said “october” or “two” (7 of 7 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten halves.”\n• serious · dates — never said “march” (7 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is three-ninths this year.”\n• minor · dates — kept the value but never said “february”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “Renewal is 02/29/2024.”\n→ it said “Renewal is zero two twenty nine twenty twenty four.”\n• fatal · large numbers — never said “one thousand forty two” (5 of 7 transcripts)\n→ we sent “Your queue position is 1,042.”\n→ it said “Jo koan posizjon is an zero katre doe.”\n• fatal · math operators — never said “three to one” (7 of 7 transcripts)\n→ we sent “Mix it at a 3:1 ratio.”\n→ it said “Mix it at a tro contra ratio.”\n• fatal · times — never said “nine” or “a m” (7 of 7 transcripts)\n→ we sent “We open at 9:00am.”\n→ it said “We openen at negeens ochtends.”\n• minor · times — kept the value but never said “six oh five”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “Boarding runs 6:05 to 6:20.”\n→ it said “Boarding runs six five to six twenty.”\n• fatal · units — never said “five foot eleven” (7 of 7 transcripts)\n→ we sent “He is 5'11\" tall.”\n→ it said “He is wiesiüksi toista tall.”"
            }
          }
        },
        {
          "provider": "Hume",
          "model": "octave-2",
          "id": "hume:octave-2",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1377",
              "s": 1377,
              "tone": "warn"
            },
            "ci": {
              "v": "29%",
              "s": 29,
              "tone": "warn",
              "ci": "24–34%"
            },
            "drift": {
              "v": "25",
              "s": 25,
              "tone": "warn"
            },
            "synth": {
              "v": "448ms",
              "s": 448,
              "lo": 448,
              "hi": 497,
              "ci": "448–497 p50–p90"
            },
            "cost": {
              "v": "$100.0",
              "s": 100,
              "tone": "warn",
              "note": "Starter tier, no pay-as-you-go rate published. Falls to $50 on Business; entry overage is $150."
            },
            "robustness": {
              "v": "0.20",
              "s": 20,
              "tone": "bad",
              "tie": 2,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (6 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of $0.99 was applied.”\n• minor · dates — said “slash” out loud; a prompt fixes this (5 of 18 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten slash zero two.”\n• minor · dates — said “slash” out loud; a prompt fixes this (7 of 23 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three slash zero nine this year.”\n• fatal · large numbers — never said “four thousand ninety six” (5 of 18 transcripts)\n→ we sent “The file is 4096 bytes.”\n→ it said “The file is four nine hundred and ninety six bytes.”\n• fatal · math operators — never said “fifteen” (5 of 7 transcripts)\n→ we sent “The area is 5×3=15 square metres.”\n→ it said “The area is five by three square meters.”\n• minor · math operators — kept the value but never said “minus”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “That leaves 10-4=6 remaining.”\n→ it said “That leaves ten to four, equals six remaining.”"
            }
          }
        },
        {
          "provider": "Qwen",
          "model": "qwen3-tts-flash",
          "id": "alibaba:qwen3-tts-flash",
          "rec": [],
          "cells": {
            "natural": {
              "v": "1310",
              "s": 1310,
              "tone": "bad"
            },
            "ci": {
              "v": "21%",
              "s": 21,
              "tone": "bad",
              "ci": "16–27%"
            },
            "drift": {
              "v": "21",
              "s": 21
            },
            "synth": {
              "v": "472ms",
              "s": 472,
              "lo": 472,
              "hi": 504,
              "ci": "472–504 p50–p90"
            },
            "cost": {
              "v": "$10.0",
              "s": 10,
              "tone": "good",
              "note": "International (Singapore) list: $0.10 per 10,000 input characters, output audio not billed (alibabacloud.com/help/en/model-studio/model-pricing). A true per-character rate, no conversion. Mainland-China scope is ~$0.115/10k."
            },
            "robustness": {
              "v": "0.50",
              "s": 50,
              "tone": "warn",
              "tie": 3,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (3 of 5 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is twelve point five dollars.”\n• fatal · currency — dropped the minus sign (5 of 5 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $9,204.59 as of today.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (5 of 5 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: 45,90 € outstanding.”\n• serious · dates — never said “january” or “first” (10 of 16 transcripts)\n→ we sent “Renewal falls on 1/1/2027.”\n→ it said “Renewal fights on Eins Einsweitausen Siemenfantsen.”\n• serious · dates — never said “october” (8 of 16 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to tennis two.”"
            }
          }
        },
        {
          "provider": "Palabra",
          "model": "palabra-tts-v1",
          "id": "palabra:palabra-tts-v1",
          "rec": [],
          "cells": {
            "natural": {
              "v": "~1520",
              "s": 1520,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "11",
              "s": 11,
              "tone": "good"
            },
            "synth": {
              "v": "72ms",
              "s": 72,
              "lo": 72,
              "hi": 79,
              "ci": "72–79 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$30.0",
              "s": 30,
              "tone": "good"
            },
            "robustness": {
              "v": "0.30",
              "s": 30,
              "tone": "bad",
              "tie": 1,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is $12.50.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of ninety-nine cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance 45 euros, 90 outstanding.”\n• minor · dates — kept the value but never said “october”; a prompt fixes this (4 of 7 transcripts)\n→ we sent “The deadline moved to 10/02.”\n→ it said “The deadline moved to ten Jessica with two.”\n• minor · dates — kept the value but never said “march”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “Your slot is 03/09 this year.”\n→ it said “Your slot is zero three to zero nine this year.”\n• fatal · decimals — never said “plus or minus” or “zero point zero five” or “millimetre” (6 of 7 transcripts)\n→ we sent “Tolerance is ±0.05 mm on that part.”\n→ it said “Tolerance is not you too zero five mon that part.”\n• minor · units — kept the value but never said “gigabyte”; a prompt fixes this (8 of 23 transcripts)\n→ we sent “You have 250GB remaining.”\n→ it said “You have two hundred and fifty GB remaining.”"
            }
          }
        },
        {
          "provider": "Bland",
          "model": "bland-speech",
          "id": "bland:bland-speech",
          "rec": [],
          "measured": "2026-08-20",
          "cells": {
            "natural": {
              "v": "~1569",
              "s": 1569,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "12",
              "s": 12,
              "tone": "good"
            },
            "synth": {
              "v": "303ms",
              "s": 303,
              "lo": 303,
              "hi": 386,
              "ci": "303–386 p50–p90"
            },
            "cost": {
              "v": "$15.0",
              "s": 15,
              "tone": "good"
            },
            "robustness": {
              "v": "0.00",
              "s": 0,
              "tone": "bad",
              "tie": 17,
              "note": "Fined:\n• fatal · currency — dropped the minus sign (5 of 5 transcripts)\n→ we sent “The balance is -$12.50.”\n→ it said “The balance is twelve dollars and fifty cents.”\n• fatal · currency — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Your balance is -$1,204.50 as of today.”\n→ it said “Your balance is $1,204.50 as of today.”\n• fatal · currency — dropped the minus sign (5 of 7 transcripts)\n→ we sent “A refund of -$0.99 was applied.”\n→ it said “A refund of ninety-nine cents was applied.”\n• fatal · currency (international, ×0.4) — dropped the minus sign (7 of 7 transcripts)\n→ we sent “Balance: -€45,90 outstanding.”\n→ it said “Balance: 45,090 outstanding.”\n• minor · dates — kept the value but never said “december”; a prompt fixes this (8 of 11 transcripts)\n→ we sent “We close on Dec 24.”\n→ it said “We close on Deck 24.”\n• fatal · decimals — dropped the minus sign (6 of 7 transcripts)\n→ we sent “Growth slowed to -0.5% this month.”\n→ it said “Growth slowed to 0.5% this month.”\n• fatal · large numbers — never said “eight point one billion” (7 of 14 transcripts)\n→ we sent “Population reached 8,100,000,000 in 2023.”\n→ it said “Population reached eight thousand one hundred thousand twenty twenty three.”\n• fatal · large numbers — never said “one million” (5 of 8 transcripts)\n→ we sent “We shipped 1,000,000 units last year.”\n→ it said “We shipped one thousand thousand units last year.”\n• fatal · large numbers — never said “eight billion” (5 of 7 transcripts)\n→ we sent “The planet now holds 8,000,000,000 people.”\n→ it said “The planet now holds eight thousand thousand people.”\n• minor · math operators — kept the value but never said “three quarters”; a prompt fixes this (3 of 4 transcripts)\n→ we sent “Give me 3/4 of the batch.”\n→ it said “Give me three-four of the batch.”\n• minor · units — kept the value but never said “miles per hour”; a prompt fixes this (3 of 7 transcripts)\n→ we sent “Speed limit is 70mph here.”\n→ it said “Speed limit is seventy mlyfres here.”\n• minor · units — kept the value but never said “kilometres per hour”; a prompt fixes this (7 of 23 transcripts)\n→ we sent “Wind is 25km/h northwest.”\n→ it said “Wind is twenty five kilometers northwest.”\n• fatal · units — dropped the minus sign (1 of 1 transcripts)\n→ we sent “The freezer sits at -18°C all day.”\n→ it said “The freezer sits at 18°C all day.”"
            }
          }
        },
        {
          "provider": "Maya",
          "model": "Maya 2 Native",
          "id": "maya:Maya 2 Native",
          "rec": [],
          "measured": "2026-08-14",
          "flag": "Indian English only — no en-US/en-GB variant, so this row is absent from the English naturalness and robustness measurements. Naturalness is on the multilingual board, Hindi column.",
          "cells": {
            "natural": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "drift": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "synth": {
              "v": "100ms",
              "s": 100,
              "lo": 100,
              "hi": 114,
              "ci": "100–114 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "~$4.0",
              "s": 4,
              "tone": "good",
              "note": "Quoted by Maya directly. They publish no rate card — no pricing page and nothing in docs.mayaresearch.ai — so unlike every other row this rate cannot be checked against a vendor page."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not measured: the NSW fixtures are en-US and Maya ships no en-US voice — its `en` is Indian English, with no en-US/en-GB variant."
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "eleven_flash_v2_5",
          "id": "elevenlabs:eleven_flash_v2_5",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: Jessica (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1494",
              "s": 1494,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "tone": "good",
              "note": "API rate card (elevenlabs.io/pricing/api, 2026-09-08): Flash and Turbo $0.05 per 1K characters; v3 and Multilingual v2 $0.10."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "eleven_turbo_v2_5",
          "id": "elevenlabs:eleven_turbo_v2_5",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: Jessica (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1466",
              "s": 1466,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "tone": "good",
              "note": "API rate card (elevenlabs.io/pricing/api, 2026-09-08): Flash and Turbo $0.05 per 1K characters; v3 and Multilingual v2 $0.10."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "eleven_multilingual_v2",
          "id": "elevenlabs:eleven_multilingual_v2",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: Jessica (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1472",
              "s": 1472,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$100",
              "s": 100,
              "tone": "warn",
              "note": "API rate card (elevenlabs.io/pricing/api, 2026-09-08): Multilingual v2 $0.10 per 1K characters, the v3 rate."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "eleven_flash_v2",
          "id": "elevenlabs:eleven_flash_v2",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: Jessica (panel voice); clips vendor-direct, the model is not in the gateway catalog. Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1507",
              "s": 1507,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "tone": "good",
              "note": "API rate card (elevenlabs.io/pricing/api, 2026-09-08): Flash and Turbo $0.05 per 1K characters."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "ElevenLabs",
          "model": "eleven_turbo_v2",
          "id": "elevenlabs:eleven_turbo_v2",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: Jessica (panel voice); clips vendor-direct, the model is not in the gateway catalog. Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1492",
              "s": 1492,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "tone": "good",
              "note": "API rate card (elevenlabs.io/pricing/api, 2026-09-08): Flash and Turbo $0.05 per 1K characters."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "Cartesia",
          "model": "sonic-3",
          "id": "cartesia:sonic-3",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: Katie (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1551",
              "s": 1551,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "note": "Derived, not published: Cartesia documents \"approximately 1 credit per character\" and Pro at $5/100K credits — the same basis as the sonic-3.5 and sonic-3.6 rows."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "Rime",
          "model": "coda",
          "id": "rime:coda",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: astra (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1535",
              "s": 1535,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$50.0",
              "s": 50,
              "note": "Starter tier, rime.ai/pricing 2026-09-08: Coda $0.05 / 1K characters."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "Rime",
          "model": "mistv3",
          "id": "rime:mistv3",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: astra (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1428",
              "s": 1428,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$30.0",
              "s": 30,
              "tone": "good",
              "note": "Starter tier, rime.ai/pricing 2026-09-08: Mist v3 $0.03 / 1K characters."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "Hume",
          "model": "octave-1",
          "id": "hume:octave-1",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: Ava Song (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1455",
              "s": 1455,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "$100.0",
              "s": 100,
              "tone": "warn",
              "note": "Hume prices Octave per character with no per-version rate; Starter tier, as on the octave-2 row."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "MiniMax",
          "model": "speech-2.6-hd",
          "id": "minimax:speech-2.6-hd",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: English_radiant_girl (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1450",
              "s": 1450,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "~$100.0",
              "s": 100,
              "tone": "warn",
              "note": "Estimate. The vendor pricing page lists point plans, not per-model rates (2026-09-08); HD taken at the tier the speech-2.8-hd row cites."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "MiniMax",
          "model": "speech-2.6-turbo",
          "id": "minimax:speech-2.6-turbo",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: English_radiant_girl (panel voice). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1408",
              "s": 1408,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "~$60.0",
              "s": 60,
              "note": "Estimate. The vendor pricing page lists point plans, not per-model rates (2026-09-08); Turbo taken at the ≈$60 the speech-2.8-hd row cites for its turbo sibling."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        },
        {
          "provider": "Smallest",
          "model": "lightning_v3.1_pro",
          "id": "smallest:lightning_v3.1_pro",
          "rec": [],
          "measured": "2026-09-08",
          "flag": "Naturalness is an auto-MOS estimate (~), not a panel vote — ±20 Elo leave-one-out, a tie not a rank. Voice: joanna (gateway default; the panel voice avery is not on the Pro tier). Synth arrives from the 2026-09-08 vendor-direct sweep.",
          "cells": {
            "natural": {
              "v": "~1447",
              "s": 1447,
              "dim": true
            },
            "ci": {
              "v": "—",
              "s": null,
              "dim": true
            },
            "cost": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "No per-model rate: smallest.ai/pricing lists a blended \"Text to Speech Layer ~$0.09/minute\" and no Pro tier price (2026-09-08)."
            },
            "robustness": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not yet measured: the en-US NSW robustness battery has not been run on this model (row added 2026-09-08 with an auto-MOS naturalness estimate). Pending, not exempt."
            }
          }
        }
      ],
      "reading": ""
    },
    {
      "id": "llm",
      "navLabel": "LLM",
      "identityLabel": "Provider / Model",
      "title": "LLM",
      "subtitle": "First-token latency, task completion, fabrication and cost for voice-agent LLMs.",
      "live": true,
      "meta": "English · corpus v3-16d3e8496362, measured 2026-08-22 to 08-25 (19 scenarios × 5 iters = n=95 task runs; 10 fabrication probes × 3 = n=30) · TTFT/cost from us-east4",
      "measured": "2026-08-25",
      "headline": [
        "ttft",
        "task",
        "cost"
      ],
      "scatter": {
        "x": "ttft",
        "y": "task",
        "cornerLabel": "fast + gets the job done"
      },
      "columns": [
        {
          "id": "score",
          "label": "Score",
          "dir": "higher",
          "info": "Retired. The board publishes no composite — each parameter is ranked on its own card."
        },
        {
          "id": "task",
          "label": "Task done",
          "dir": "higher",
          "info": "Mean per-run score over 19 scripted scenarios: 1 completed, 0 acted dangerously, part-credit if it stopped early but safely."
        },
        {
          "id": "refusal",
          "label": "Refusal quality",
          "dir": "higher",
          "info": "When it does not know, whether it names the gap and offers a way forward. Higher is better."
        },
        {
          "id": "fabric",
          "label": "Fabrication",
          "dir": "lower",
          "info": "Invented a policy or fact it was never given, % of out-of-policy questions."
        },
        {
          "id": "deadair",
          "label": "Dead-air",
          "dir": "lower",
          "info": "Turns where the caller hears silence (no spoken words), % of turns."
        },
        {
          "id": "toolsilence",
          "label": "Tool silence",
          "dir": "lower",
          "info": "Tool-call turns with no spoken lead-in, % — the caller waits in silence. Not in the score."
        },
        {
          "id": "stall",
          "label": "Stalled",
          "dir": "lower",
          "info": "Ended a turn mid-task with no tool call — the caller waits and has to re-ask, % of runs."
        },
        {
          "id": "ttft",
          "label": "TTFT p50",
          "dir": "lower",
          "info": "Time-to-first-token, p50 (us-east4). Whisker spans p50→p90."
        },
        {
          "id": "cost",
          "label": "Cost / 1M tok",
          "dir": "lower",
          "info": "$ per 1M output tokens (host / vendor list pricing)."
        }
      ],
      "rows": [
        {
          "provider": "Baseten",
          "model": "GLM-4.7",
          "id": "baseten:zai-org/GLM-4.7",
          "meta": "reasoning off (gateway default)",
          "measured": "2026-08-17",
          "status": {
            "label": "STALLS",
            "tone": "warn"
          },
          "flag": "Quietest of the group, but ends a turn mid-task on 43.3% of runs.",
          "rec": [],
          "cells": {
            "score": {
              "v": "89",
              "s": 89,
              "tone": "good"
            },
            "stall": {
              "v": "43.3%",
              "s": 43.3,
              "tone": "bad",
              "ci": "39 of 90 runs"
            },
            "fabric": {
              "v": "17%",
              "s": 17,
              "tone": "warn",
              "ci": "5 of 30 probe runs"
            },
            "deadair": {
              "v": "2.3%",
              "s": 2.3,
              "tone": "good",
              "ci": "6 of 260 turns"
            },
            "toolsilence": {
              "v": "3.7%",
              "s": 3.7,
              "tone": "good",
              "ci": "6 of 161 tool-call turns"
            },
            "task": {
              "v": "72.46%",
              "s": 72.46,
              "tone": "warn",
              "ci": "59–85% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "80.0%",
              "s": 80,
              "tone": "good",
              "ci": "names the gap 67% · offers a route 83% · 10 probes x 3 iterations"
            },
            "ttft": {
              "v": "275ms",
              "s": 275,
              "lo": 275,
              "hi": 431,
              "ci": "275–431 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$2.20",
              "s": 2.2
            }
          },
          "lead": true
        },
        {
          "provider": "Baseten",
          "model": "DeepSeek-V4-Flash-0731",
          "id": "baseten:deepseek-ai/DeepSeek-V4-Flash-0731",
          "meta": "reasoning off (gateway default)",
          "measured": "2026-08-17",
          "status": {
            "label": "STALLS",
            "tone": "warn"
          },
          "flag": "Cheapest row on the board; stalls mid-task on 27.8% of runs.",
          "rec": [],
          "cells": {
            "score": {
              "v": "89",
              "s": 89,
              "tone": "good"
            },
            "stall": {
              "v": "27.8%",
              "s": 27.8,
              "tone": "warn",
              "ci": "25 of 90 runs"
            },
            "fabric": {
              "v": "10%",
              "s": 10,
              "tone": "good",
              "ci": "3 of 30 probe runs"
            },
            "deadair": {
              "v": "5.1%",
              "s": 5.1,
              "tone": "good",
              "ci": "16 of 314 turns"
            },
            "toolsilence": {
              "v": "7.4%",
              "s": 7.4,
              "tone": "good",
              "ci": "16 of 217 tool-call turns"
            },
            "task": {
              "v": "80.18%",
              "s": 80.18,
              "tone": "good",
              "ci": "71–90% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "72.2%",
              "s": 72.2,
              "tone": "warn",
              "ci": "names the gap 57% · offers a route 67% · 10 probes x 3 iterations"
            },
            "ttft": {
              "v": "361ms",
              "s": 361,
              "lo": 361,
              "hi": 432,
              "ci": "361–432 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$0.26",
              "s": 0.26
            }
          },
          "lead": false
        },
        {
          "provider": "Anthropic",
          "model": "Claude Haiku 4.5",
          "id": "anthropic:claude-haiku-4-5",
          "status": {
            "label": "STRONG",
            "tone": "good"
          },
          "flag": "Never fabricated, near-zero dead-air — 532ms TTFT.",
          "rec": [
            {
              "label": "Live agent"
            }
          ],
          "cells": {
            "score": {
              "v": "88",
              "s": 88,
              "tone": "good"
            },
            "stall": {
              "v": "25.6%",
              "s": 25.6,
              "tone": "warn",
              "ci": "23 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "1.6%",
              "s": 1.6,
              "tone": "good",
              "ci": "4 of 248 turns"
            },
            "toolsilence": {
              "v": "2.7%",
              "s": 2.7,
              "tone": "good",
              "ci": "4 of 148 tool-call turns"
            },
            "task": {
              "v": "82.46%",
              "s": 82.46,
              "tone": "good",
              "ci": "67–95% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "84.4%",
              "s": 84.4,
              "tone": "good",
              "ci": "names the gap 77% · offers a route 80% · 10 probes x 3 iterations"
            },
            "toolarg": {
              "v": "77%",
              "s": 77,
              "tone": "bad"
            },
            "placeholder": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "ttft": {
              "v": "532ms",
              "s": 532,
              "lo": 532,
              "hi": 1787,
              "ci": "532–1787 p50–p90 · us-east4 direct",
              "tone": "good"
            },
            "cost": {
              "v": "$5.00",
              "s": 5
            }
          },
          "lead": false
        },
        {
          "provider": "Cerebras",
          "model": "gemma-4-31b",
          "id": "cerebras:gemma-4-31b",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "flag": "Stalls after the lookup on 38.9% of runs — announces the refund, never issues it.",
          "rec": [
            {
              "label": "Low latency"
            }
          ],
          "cells": {
            "score": {
              "v": "84",
              "s": 84,
              "tone": "good"
            },
            "stall": {
              "v": "38.9%",
              "s": 38.9,
              "tone": "bad",
              "ci": "35 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "53.3%",
              "s": 53.3,
              "tone": "bad",
              "ci": "122 of 229 turns"
            },
            "toolsilence": {
              "v": "94.6%",
              "s": 94.6,
              "tone": "bad",
              "ci": "122 of 129 tool-call turns"
            },
            "task": {
              "v": "77.72%",
              "s": 77.72,
              "tone": "warn",
              "ci": "62–92% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "90.0%",
              "s": 90,
              "tone": "good",
              "ci": "names the gap 87% · offers a route 90% · 10 probes x 3 iterations · 2 silent turns scored zero"
            },
            "toolarg": {
              "v": "33%",
              "s": 33,
              "tone": "bad"
            },
            "placeholder": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "ttft": {
              "v": "192ms",
              "s": 192,
              "lo": 192,
              "hi": 232,
              "ci": "192–232 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$1.49",
              "s": 1.49
            }
          },
          "lead": false
        },
        {
          "provider": "Cerebras",
          "model": "gpt-oss-120b",
          "id": "cerebras:gpt-oss-120b",
          "status": {
            "label": "DEAD AIR",
            "tone": "warn"
          },
          "flag": "Silent on 100% of tool calls — dead air on every action.",
          "rec": [
            {
              "label": "Needs bridging",
              "caution": true
            }
          ],
          "cells": {
            "score": {
              "v": "80",
              "s": 80,
              "tone": "good"
            },
            "stall": {
              "v": "15.6%",
              "s": 15.6,
              "tone": "warn",
              "ci": "14 of 90 runs"
            },
            "fabric": {
              "v": "30%",
              "s": 30,
              "tone": "warn",
              "ci": "9 of 30 probe runs"
            },
            "deadair": {
              "v": "68.2%",
              "s": 68.2,
              "tone": "bad",
              "ci": "204 of 299 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (204 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "85.00%",
              "s": 85,
              "tone": "good",
              "ci": "70–96% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "83.3%",
              "s": 83.3,
              "tone": "good",
              "ci": "names the gap 77% · offers a route 83% · 10 probes x 3 iterations · 3 silent turns scored zero"
            },
            "toolarg": {
              "v": "50%",
              "s": 50,
              "tone": "bad"
            },
            "placeholder": {
              "v": "13%",
              "s": 13,
              "tone": "bad"
            },
            "ttft": {
              "v": "195ms",
              "s": 195,
              "lo": 195,
              "hi": 260,
              "ci": "195–260 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$0.75",
              "s": 0.75,
              "tone": "good"
            }
          },
          "lead": false
        },
        {
          "provider": "Baseten",
          "model": "inkling",
          "id": "baseten:thinkingmachines/inkling",
          "meta": "reasoning off (gateway default)",
          "measured": "2026-08-17",
          "status": {
            "label": "WEAK",
            "tone": "bad"
          },
          "flag": "975B total for 41B active — 211ms first token, but stalls on 42.2% of runs.",
          "rec": [],
          "cells": {
            "score": {
              "v": "79",
              "s": 79,
              "tone": "warn"
            },
            "stall": {
              "v": "42.2%",
              "s": 42.2,
              "tone": "bad",
              "ci": "38 of 90 runs"
            },
            "fabric": {
              "v": "3%",
              "s": 3,
              "tone": "good",
              "ci": "1 of 30 probe runs"
            },
            "deadair": {
              "v": "66.3%",
              "s": 66.3,
              "tone": "bad",
              "ci": "203 of 306 turns"
            },
            "toolsilence": {
              "v": "94.0%",
              "s": 94,
              "tone": "bad",
              "ci": "203 of 216 tool-call turns"
            },
            "task": {
              "v": "71.14%",
              "s": 71.14,
              "tone": "warn",
              "ci": "57–84% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "74.4%",
              "s": 74.4,
              "tone": "warn",
              "ci": "names the gap 60% · offers a route 73% · 10 probes x 3 iterations · 3 silent turns scored zero"
            },
            "ttft": {
              "v": "211ms",
              "s": 211,
              "lo": 211,
              "hi": 342,
              "ci": "211–342 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$4.05",
              "s": 4.05
            }
          },
          "lead": false
        },
        {
          "provider": "Baseten",
          "model": "Nemotron-3-Ultra",
          "id": "baseten:nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
          "meta": "non-thinking at default",
          "measured": "2026-08-18",
          "status": {
            "label": "STALLS",
            "tone": "warn"
          },
          "flag": "Fabricates on 27% of out-of-policy questions and stalls on 54.4% of runs.",
          "rec": [],
          "cells": {
            "score": {
              "v": "76",
              "s": 76,
              "tone": "warn"
            },
            "stall": {
              "v": "54.4%",
              "s": 54.4,
              "tone": "bad",
              "ci": "49 of 90 runs"
            },
            "fabric": {
              "v": "27%",
              "s": 27,
              "tone": "warn",
              "ci": "8 of 30 probe runs"
            },
            "deadair": {
              "v": "56.0%",
              "s": 56,
              "tone": "bad",
              "ci": "139 of 248 turns"
            },
            "toolsilence": {
              "v": "93.9%",
              "s": 93.9,
              "tone": "bad",
              "ci": "139 of 148 tool-call turns"
            },
            "task": {
              "v": "72.54%",
              "s": 72.54,
              "tone": "warn",
              "ci": "60–86% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "75.6%",
              "s": 75.6,
              "tone": "warn",
              "ci": "names the gap 83% · offers a route 60% · 10 probes x 3 iterations · 1 silent turn scored zero"
            },
            "ttft": {
              "v": "300ms",
              "s": 300,
              "lo": 300,
              "hi": 370,
              "ci": "300–370 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$2.40",
              "s": 2.4
            }
          },
          "lead": false
        },
        {
          "provider": "Baseten",
          "model": "gpt-oss-120b",
          "id": "baseten:openai/gpt-oss-120b",
          "meta": "reasoning off (gateway default)",
          "measured": "2026-08-17",
          "status": {
            "label": "DEAD AIR",
            "tone": "warn"
          },
          "flag": "Silent on every tool call; fabricates on 27% of out-of-policy questions.",
          "rec": [],
          "cells": {
            "score": {
              "v": "75",
              "s": 75,
              "tone": "warn"
            },
            "stall": {
              "v": "8.9%",
              "s": 8.9,
              "tone": "good",
              "ci": "8 of 90 runs"
            },
            "fabric": {
              "v": "27%",
              "s": 27,
              "tone": "warn",
              "ci": "8 of 30 probe runs"
            },
            "deadair": {
              "v": "71.2%",
              "s": 71.2,
              "tone": "bad",
              "ci": "227 of 319 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (227 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "86.32%",
              "s": 86.32,
              "tone": "good",
              "ci": "72–98% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "80.0%",
              "s": 80,
              "tone": "good",
              "ci": "names the gap 67% · offers a route 83% · 10 probes x 3 iterations · 3 silent turns scored zero"
            },
            "ttft": {
              "v": "393ms",
              "s": 393,
              "lo": 393,
              "hi": 536,
              "ci": "393–536 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$0.50",
              "s": 0.5
            }
          },
          "lead": false
        },
        {
          "provider": "Baseten",
          "model": "inkling-small",
          "id": "baseten:thinkingmachines/inkling-small",
          "meta": "reasoning off (gateway default)",
          "measured": "2026-08-17",
          "status": {
            "label": "WEAK",
            "tone": "bad"
          },
          "flag": "Fastest first token measured here; repeated a state-changing refund on 5 of 5 H3 runs.",
          "rec": [],
          "cells": {
            "score": {
              "v": "75",
              "s": 75,
              "tone": "warn"
            },
            "stall": {
              "v": "45.6%",
              "s": 45.6,
              "tone": "bad",
              "ci": "41 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "72.0%",
              "s": 72,
              "tone": "bad",
              "ci": "244 of 339 turns"
            },
            "toolsilence": {
              "v": "94.9%",
              "s": 94.9,
              "tone": "bad",
              "ci": "244 of 257 tool-call turns"
            },
            "task": {
              "v": "57.72%",
              "s": 57.72,
              "tone": "warn",
              "ci": "43–73% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "77.8%",
              "s": 77.8,
              "tone": "warn",
              "ci": "names the gap 77% · offers a route 57% · 10 probes x 3 iterations"
            },
            "ttft": {
              "v": "177ms",
              "s": 177,
              "lo": 177,
              "hi": 268,
              "ci": "177–268 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$1.20",
              "s": 1.2
            }
          },
          "lead": false
        },
        {
          "provider": "OpenAI",
          "model": "gpt-5.6-luna",
          "id": "openai:gpt-5.6-luna",
          "meta": "reasoning off (gateway default)",
          "measured": "2026-08-05",
          "status": {
            "label": "SOLID",
            "tone": "good"
          },
          "flag": "Re-measured 2026-08-05 with reasoning off (0 reasoning tokens).",
          "rec": [],
          "cells": {
            "score": {
              "v": "74",
              "s": 74,
              "tone": "warn"
            },
            "stall": {
              "v": "7.8%",
              "s": 7.8,
              "tone": "good",
              "ci": "7 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "69.7%",
              "s": 69.7,
              "tone": "bad",
              "ci": "230 of 330 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (230 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "94.21%",
              "s": 94.21,
              "tone": "good",
              "ci": "84–100% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "95.6%",
              "s": 95.6,
              "tone": "good",
              "ci": "names the gap 97% · offers a route 93% · 10 probes x 3 iterations"
            },
            "toolarg": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "placeholder": {
              "v": "93%",
              "s": 93,
              "tone": "bad"
            },
            "ttft": {
              "v": "659ms",
              "s": 659,
              "lo": 659,
              "hi": 849,
              "ci": "659–849 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$1.20",
              "s": 1.2,
              "tone": "good"
            }
          },
          "lead": false
        },
        {
          "provider": "Gemini",
          "model": "gemini-3.5-flash-lite",
          "id": "gemini:gemini-3.5-flash-lite",
          "meta": "reasoning at minimal (family floor)",
          "status": {
            "label": "SOLID",
            "tone": "good"
          },
          "flag": "Measured 2026-07-25; drops some compound multi-step requests.",
          "rec": [],
          "cells": {
            "score": {
              "v": "74",
              "s": 74,
              "tone": "warn"
            },
            "stall": {
              "v": "27.8%",
              "s": 27.8,
              "tone": "warn",
              "ci": "25 of 90 runs"
            },
            "fabric": {
              "v": "10%",
              "s": 10,
              "tone": "good",
              "ci": "3 of 30 probe runs"
            },
            "deadair": {
              "v": "68.6%",
              "s": 68.6,
              "tone": "bad",
              "ci": "197 of 287 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (197 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "72.02%",
              "s": 72.02,
              "tone": "warn",
              "ci": "53–89% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "70.0%",
              "s": 70,
              "tone": "warn",
              "ci": "names the gap 63% · offers a route 70% · 10 probes x 3 iterations · 6 silent turns scored zero"
            },
            "toolarg": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "placeholder": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "ttft": {
              "v": "373ms",
              "s": 373,
              "lo": 373,
              "hi": 404,
              "ci": "373–404 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$2.50",
              "s": 2.5
            }
          },
          "lead": false
        },
        {
          "provider": "OpenAI",
          "model": "gpt-4.1",
          "id": "openai:gpt-4.1",
          "status": {
            "label": "STRONG",
            "tone": "good"
          },
          "flag": "Fast enough for a live turn; priciest of the group. TTFT measured 2026-07-27.",
          "rec": [
            {
              "label": "Live agent"
            }
          ],
          "cells": {
            "score": {
              "v": "74",
              "s": 74,
              "tone": "warn"
            },
            "stall": {
              "v": "14.4%",
              "s": 14.4,
              "tone": "warn",
              "ci": "13 of 90 runs"
            },
            "fabric": {
              "v": "10%",
              "s": 10,
              "tone": "good",
              "ci": "3 of 30 probe runs"
            },
            "deadair": {
              "v": "58.9%",
              "s": 58.9,
              "tone": "bad",
              "ci": "159 of 270 turns"
            },
            "toolsilence": {
              "v": "93.5%",
              "s": 93.5,
              "tone": "bad",
              "ci": "159 of 170 tool-call turns"
            },
            "task": {
              "v": "91.40%",
              "s": 91.4,
              "tone": "good",
              "ci": "78–100% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "78.9%",
              "s": 78.9,
              "tone": "warn",
              "ci": "names the gap 70% · offers a route 80% · 10 probes x 3 iterations · 3 silent turns scored zero"
            },
            "toolarg": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "placeholder": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "ttft": {
              "v": "640ms",
              "s": 640,
              "lo": 640,
              "hi": 775,
              "ci": "640–775 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$8.00",
              "s": 8,
              "tone": "warn"
            }
          },
          "lead": false
        },
        {
          "provider": "OpenAI",
          "model": "gpt-5.6-terra",
          "id": "openai:gpt-5.6-terra",
          "meta": "reasoning off (gateway default)",
          "measured": "2026-08-05",
          "status": {
            "label": "SOLID",
            "tone": "good"
          },
          "flag": "Measured 2026-08-05 with reasoning off, in the same run as luna.",
          "rec": [],
          "cells": {
            "score": {
              "v": "71",
              "s": 71,
              "tone": "warn"
            },
            "stall": {
              "v": "17.8%",
              "s": 17.8,
              "tone": "warn",
              "ci": "16 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "66.3%",
              "s": 66.3,
              "tone": "bad",
              "ci": "195 of 294 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (195 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "87.37%",
              "s": 87.37,
              "tone": "good",
              "ci": "74–98% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "88.9%",
              "s": 88.9,
              "tone": "good",
              "ci": "names the gap 90% · offers a route 87% · 10 probes x 3 iterations · 1 silent turn scored zero"
            },
            "toolarg": {
              "v": "7%",
              "s": 7,
              "tone": "warn"
            },
            "placeholder": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "ttft": {
              "v": "701ms",
              "s": 701,
              "lo": 701,
              "hi": 797,
              "ci": "701–797 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$12.00",
              "s": 12,
              "tone": "warn"
            }
          },
          "lead": false
        },
        {
          "provider": "OpenAI",
          "model": "gpt-5-mini",
          "id": "openai:gpt-5-mini",
          "meta": "reasoning",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "score": {
              "v": "71",
              "s": 71,
              "tone": "warn"
            },
            "stall": {
              "v": "21.1%",
              "s": 21.1,
              "tone": "warn",
              "ci": "19 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "67.1%",
              "s": 67.1,
              "tone": "bad",
              "ci": "198 of 295 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (198 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "89.12%",
              "s": 89.12,
              "tone": "good",
              "ci": "82–95% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "83.3%",
              "s": 83.3,
              "tone": "good",
              "ci": "names the gap 83% · offers a route 83% · 10 probes x 3 iterations · 5 silent turns scored zero"
            },
            "toolarg": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "placeholder": {
              "v": "7%",
              "s": 7,
              "tone": "warn"
            },
            "ttft": {
              "v": "746ms",
              "s": 746,
              "lo": 746,
              "hi": 794,
              "ci": "746–794 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$2.00",
              "s": 2
            }
          },
          "lead": false
        },
        {
          "provider": "OpenAI",
          "model": "gpt-4.1-mini",
          "id": "openai:gpt-4.1-mini",
          "status": {
            "label": "FABRICATES",
            "tone": "warn"
          },
          "flag": "Fabricates on 57% of out-of-policy questions — the worst rate on the board bar qwen-turbo. TTFT measured 2026-07-27.",
          "rec": [
            {
              "label": "Fast + capable",
              "caution": true
            }
          ],
          "cells": {
            "score": {
              "v": "70",
              "s": 70,
              "tone": "warn"
            },
            "stall": {
              "v": "16.7%",
              "s": 16.7,
              "tone": "warn",
              "ci": "15 of 90 runs"
            },
            "fabric": {
              "v": "57%",
              "s": 57,
              "tone": "bad",
              "ci": "17 of 30 probe runs"
            },
            "deadair": {
              "v": "40.0%",
              "s": 40,
              "tone": "bad",
              "ci": "102 of 255 turns"
            },
            "toolsilence": {
              "v": "65.8%",
              "s": 65.8,
              "tone": "bad",
              "ci": "102 of 155 tool-call turns"
            },
            "task": {
              "v": "87.19%",
              "s": 87.19,
              "tone": "good",
              "ci": "78–95% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "83.3%",
              "s": 83.3,
              "tone": "good",
              "ci": "names the gap 80% · offers a route 80% · 10 probes x 3 iterations · 3 silent turns scored zero"
            },
            "toolarg": {
              "v": "27%",
              "s": 27,
              "tone": "bad"
            },
            "placeholder": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "ttft": {
              "v": "708ms",
              "s": 708,
              "lo": 708,
              "hi": 1094,
              "ci": "708–1094 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$1.60",
              "s": 1.6,
              "tone": "good"
            }
          },
          "lead": false
        },
        {
          "provider": "OpenAI",
          "model": "gpt-5",
          "id": "openai:gpt-5",
          "meta": "reasoning",
          "status": {
            "label": "SOLID",
            "tone": "good"
          },
          "rec": [],
          "cells": {
            "score": {
              "v": "67",
              "s": 67,
              "tone": "warn"
            },
            "stall": {
              "v": "25.6%",
              "s": 25.6,
              "tone": "warn",
              "ci": "23 of 90 runs"
            },
            "fabric": {
              "v": "10%",
              "s": 10,
              "tone": "good",
              "ci": "3 of 30 probe runs"
            },
            "deadair": {
              "v": "71.8%",
              "s": 71.8,
              "tone": "bad",
              "ci": "216 of 301 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (216 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "74.74%",
              "s": 74.74,
              "tone": "warn",
              "ci": "60–88% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "84.4%",
              "s": 84.4,
              "tone": "good",
              "ci": "names the gap 77% · offers a route 77% · 10 probes x 3 iterations"
            },
            "toolarg": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "placeholder": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "ttft": {
              "v": "635ms",
              "s": 635,
              "lo": 635,
              "hi": 679,
              "ci": "635–679 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$10.00",
              "s": 10,
              "tone": "warn"
            }
          },
          "lead": false
        },
        {
          "provider": "OpenAI",
          "model": "gpt-5-nano",
          "id": "openai:gpt-5-nano",
          "meta": "reasoning",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "score": {
              "v": "64",
              "s": 64,
              "tone": "warn"
            },
            "stall": {
              "v": "86.7%",
              "s": 86.7,
              "tone": "bad",
              "ci": "78 of 90 runs"
            },
            "fabric": {
              "v": "17%",
              "s": 17,
              "tone": "warn",
              "ci": "5 of 30 probe runs"
            },
            "deadair": {
              "v": "39.4%",
              "s": 39.4,
              "tone": "bad",
              "ci": "65 of 165 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 17 items (65 tool-call turns) · ≥80%"
            },
            "task": {
              "v": "32.37%",
              "s": 32.37,
              "tone": "bad",
              "ci": "21–43% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "84.4%",
              "s": 84.4,
              "tone": "good",
              "ci": "names the gap 77% · offers a route 77% · 10 probes x 3 iterations"
            },
            "toolarg": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "placeholder": {
              "v": "23%",
              "s": 23,
              "tone": "bad"
            },
            "ttft": {
              "v": "525ms",
              "s": 525,
              "lo": 525,
              "hi": 1020,
              "ci": "525–1020 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$0.40",
              "s": 0.4,
              "tone": "good"
            }
          },
          "lead": false
        },
        {
          "provider": "Together",
          "model": "Llama-3.3-70B",
          "id": "together:meta-llama/Llama-3.3-70B-Instruct-Turbo",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "flag": "Fails restraint — fires a tool with a made-up ID instead of asking for the missing one.",
          "rec": [],
          "cells": {
            "score": {
              "v": "63",
              "s": 63,
              "tone": "warn"
            },
            "stall": {
              "v": "26.7%",
              "s": 26.7,
              "tone": "warn",
              "ci": "24 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "74.3%",
              "s": 74.3,
              "tone": "bad",
              "ci": "260 of 350 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (260 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "64.21%",
              "s": 64.21,
              "tone": "warn",
              "ci": "43–84% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "20.0%",
              "s": 20,
              "tone": "bad",
              "ci": "names the gap 33% · offers a route 13% · 10 probes x 3 iterations · 19 silent turns scored zero"
            },
            "toolarg": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "placeholder": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "ttft": {
              "v": "698ms",
              "s": 698,
              "lo": 698,
              "hi": 860,
              "ci": "698–860 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$1.04",
              "s": 1.04,
              "tone": "good"
            }
          },
          "lead": false
        },
        {
          "provider": "Anthropic",
          "model": "Claude Sonnet 5",
          "id": "anthropic:claude-sonnet-5",
          "status": {
            "label": "SOLID",
            "tone": "good"
          },
          "flag": "1.21s TTFT — too slow to lead a live turn.",
          "rec": [],
          "cells": {
            "score": {
              "v": "62",
              "s": 62,
              "tone": "warn"
            },
            "stall": {
              "v": "22.7%",
              "s": 22.7,
              "tone": "warn",
              "ci": "20 of 88 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "39.8%",
              "s": 39.8,
              "tone": "bad",
              "ci": "117 of 294 turns"
            },
            "toolsilence": {
              "v": "58.8%",
              "s": 58.8,
              "tone": "bad",
              "ci": "117 of 199 tool-call turns"
            },
            "task": {
              "v": "81.72%",
              "s": 81.72,
              "tone": "good",
              "ci": "65–95% 95% CI bootstrap over items · n=93 · measured 2026-08-25"
            },
            "refusal": {
              "v": "90.0%",
              "s": 90,
              "tone": "good",
              "ci": "names the gap 77% · offers a route 93% · 10 probes x 3 iterations"
            },
            "toolarg": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "placeholder": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "ttft": {
              "v": "1.21s",
              "s": 1206,
              "lo": 1206,
              "hi": 1373,
              "ci": "1.21–1.37s p50–p90 · us-east4 direct",
              "tone": "warn"
            },
            "cost": {
              "v": "$10.00",
              "s": 10,
              "tone": "warn",
              "note": "Introductory rate through 2026-08-31; $15.00 from 2026-09-01."
            }
          },
          "lead": false
        },
        {
          "provider": "Alibaba",
          "model": "qwen-turbo",
          "id": "alibaba:qwen-turbo",
          "status": {
            "label": "WEAK",
            "tone": "bad"
          },
          "flag": "Promises to act and then goes quiet on 94.4% of runs; fabricates on 87% of out-of-policy questions.",
          "rec": [],
          "cells": {
            "score": {
              "v": "56",
              "s": 56,
              "tone": "bad"
            },
            "stall": {
              "v": "94.4%",
              "s": 94.4,
              "tone": "bad",
              "ci": "85 of 90 runs"
            },
            "fabric": {
              "v": "87%",
              "s": 87,
              "tone": "bad",
              "ci": "26 of 30 probe runs"
            },
            "deadair": {
              "v": "9.1%",
              "s": 9.1,
              "tone": "good",
              "ci": "10 of 110 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 2 items (10 tool-call turns) · ≥16%",
              "note": "only 10 tool-call turns — the model rarely calls a tool"
            },
            "task": {
              "v": "6.58%",
              "s": 6.58,
              "tone": "bad",
              "ci": "0–18% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "11.1%",
              "s": 11.1,
              "tone": "bad",
              "ci": "names the gap 0% · offers a route 10% · 10 probes x 3 iterations · 3 silent turns scored zero"
            },
            "toolarg": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "placeholder": {
              "v": "73%",
              "s": 73,
              "tone": "bad"
            },
            "ttft": {
              "v": "483ms",
              "s": 483,
              "lo": 483,
              "hi": 497,
              "ci": "483–497 p50–p90",
              "tone": "good"
            },
            "cost": {
              "v": "$0.20",
              "s": 0.2,
              "tone": "good"
            }
          },
          "lead": false
        },
        {
          "provider": "xAI",
          "model": "Grok-4.3",
          "id": "xai:grok-4.3",
          "status": {
            "label": "SLOW",
            "tone": "warn"
          },
          "flag": "2.15s TTFT — too slow for a live turn.",
          "rec": [],
          "cells": {
            "score": {
              "v": "39",
              "s": 39,
              "tone": "bad"
            },
            "stall": {
              "v": "6.7%",
              "s": 6.7,
              "tone": "good",
              "ci": "6 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "54.8%",
              "s": 54.8,
              "tone": "bad",
              "ci": "155 of 283 turns"
            },
            "toolsilence": {
              "v": "83.3%",
              "s": 83.3,
              "tone": "bad",
              "ci": "155 of 186 tool-call turns"
            },
            "task": {
              "v": "92.46%",
              "s": 92.46,
              "tone": "good",
              "ci": "83–99% 95% CI bootstrap over items · n=95 · measured 2026-08-25"
            },
            "refusal": {
              "v": "96.7%",
              "s": 96.7,
              "tone": "good",
              "ci": "names the gap 97% · offers a route 97% · 10 probes x 3 iterations · 1 silent turn scored zero"
            },
            "toolarg": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "placeholder": {
              "v": "20%",
              "s": 20,
              "tone": "bad"
            },
            "ttft": {
              "v": "2.15s",
              "s": 2150,
              "lo": 2150,
              "hi": 2640,
              "ci": "2.15–2.64s p50–p90",
              "tone": "warn"
            },
            "cost": {
              "v": "$2.50",
              "s": 2.5
            }
          },
          "lead": false
        },
        {
          "provider": "Together",
          "model": "DeepSeek-V4-Pro",
          "id": "together:deepseek-ai/DeepSeek-V4-Pro",
          "status": {
            "label": "NOT REAL-TIME",
            "tone": "warn"
          },
          "flag": "1.6T reasoning model with no gateway latency measurement; the regional cards carry one, reasoning off.",
          "rec": [
            {
              "label": "Quality ceiling",
              "caution": true
            }
          ],
          "cells": {
            "score": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "No real-time first-token measurement, so a speed-weighted score is not computed — unranked rather than judged on partial data."
            },
            "stall": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured on the current corpus"
            },
            "fabric": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "fabrication probes not run for this model"
            },
            "deadair": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured on the current corpus"
            },
            "toolsilence": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "no tool-call turns to measure"
            },
            "task": {
              "v": "—",
              "s": null,
              "tone": "warn",
              "ci": "no measurement on the current corpus — transport unavailable"
            },
            "refusal": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not measured on the current probe set — this row has no probe replies to score."
            },
            "toolarg": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "placeholder": {
              "v": "3%",
              "s": 3,
              "tone": "warn"
            },
            "ttft": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "No real-time first-token measurement; exceeds the ~800ms voice latency budget."
            },
            "cost": {
              "v": "$3.48",
              "s": 3.48
            }
          },
          "lead": false
        },
        {
          "provider": "Gemini",
          "model": "gemini-3.8-flash",
          "id": "gemini:gemini-3.8-flash",
          "measured": "2026-09-03",
          "status": {
            "label": "UNSAFE ACTIONS",
            "tone": "warn"
          },
          "flag": "Measured 2026-09-03; forbids-and-duplicates a state-changing cancel on two scenarios, 5/5 iterations each. Fabricates nothing (0/30).",
          "rec": [],
          "cells": {
            "score": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "No real-time first-token measurement, so a speed-weighted score is not computed — unranked rather than judged on partial data."
            },
            "stall": {
              "v": "28.9%",
              "s": 28.9,
              "tone": "warn",
              "ci": "26 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "69.0%",
              "s": 69,
              "tone": "bad",
              "ci": "214 of 310 turns"
            },
            "toolsilence": {
              "v": "100.0%",
              "s": 100,
              "tone": "bad",
              "ci": "all of 19 items (214 tool-call turns) · ≥82%"
            },
            "task": {
              "v": "77.37%",
              "s": 77.37,
              "tone": "warn",
              "ci": "61–92% 95% CI bootstrap over items · n=95 · measured 2026-09-03"
            },
            "refusal": {
              "v": "96.7%",
              "s": 96.7,
              "tone": "good",
              "ci": "names the gap 97% · offers a route 97% · 10 probes x 3 iterations"
            },
            "toolarg": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured for this model"
            },
            "placeholder": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured for this model"
            },
            "ttft": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Bimodal: about half the calls answer near 600ms and the rest take 1-5s, so the median lands in the gap and shifts between sweeps (668ms, then 1168ms minutes later). Unresolved on both transports and in both regions."
            },
            "cost": {
              "v": "$3.75",
              "s": 3.75,
              "note": "Vendor list, paid tier, per 1M output tokens; rises to $7.50 on 2027-01-01."
            }
          },
          "lead": false
        },
        {
          "provider": "LiveKit",
          "model": "kimi-k2.5",
          "id": "livekit:moonshotai/kimi-k2.5",
          "meta": "worker transport",
          "status": {
            "label": "RETIRED",
            "tone": "warn"
          },
          "flag": "Retired by LiveKit 2026-07-23. Thinking serving (not disableable) — 1.39s to first spoken token; measured 2026-07-27 on the LiveKit gateway.",
          "rec": [],
          "cells": {
            "score": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "No real-time first-token measurement, so a speed-weighted score is not computed — unranked rather than judged on partial data."
            },
            "stall": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured on the current corpus"
            },
            "fabric": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "fabrication probes not run for this model"
            },
            "deadair": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured on the current corpus"
            },
            "toolsilence": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "no tool-call turns to measure"
            },
            "task": {
              "v": "—",
              "s": null,
              "tone": "warn",
              "ci": "no measurement on the current corpus — transport unavailable"
            },
            "refusal": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not measured on the current probe set — this row has no probe replies to score."
            },
            "toolarg": {
              "v": "0%",
              "s": 0,
              "tone": "good"
            },
            "placeholder": {
              "v": "7%",
              "s": 7,
              "tone": "warn"
            },
            "ttft": {
              "v": "1.39s",
              "s": 1393,
              "lo": 1393,
              "hi": 2812,
              "ci": "1.39–2.81s p50–p90",
              "tone": "warn"
            },
            "cost": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "LiveKit sunset kimi-k2.5 on 2026-07-23 and publishes no rate for it. Replacement Kimi K2.6 is $4.00 / 1M output."
            }
          },
          "lead": false
        },
        {
          "provider": "Cerebras",
          "model": "Qwen3.8-27B",
          "id": "cerebras:qwen-3.8-27b",
          "meta": "gateway · reasoning low (gateway default)",
          "status": {
            "label": "SOLID",
            "tone": "good"
          },
          "flag": "Same weights as the Alibaba Qwen3.8-27B row on a different host and at a different reasoning setting, so read the pair as host+setting, not host alone.",
          "rec": [],
          "cells": {
            "score": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "No real-time first-token measurement, so a speed-weighted score is not computed — unranked rather than judged on partial data."
            },
            "stall": {
              "v": "14.4%",
              "s": 14.4,
              "tone": "warn",
              "ci": "13 of 90 runs"
            },
            "fabric": {
              "v": "0%",
              "s": 0,
              "tone": "good",
              "ci": "0 of 10 probes (30 probe runs) · ≤31%"
            },
            "deadair": {
              "v": "33.2%",
              "s": 33.2,
              "tone": "bad",
              "ci": "98 of 295 turns"
            },
            "toolsilence": {
              "v": "48.7%",
              "s": 48.7,
              "tone": "bad",
              "ci": "97 of 199 tool-call turns"
            },
            "task": {
              "v": "87.72%",
              "s": 87.72,
              "tone": "good",
              "ci": "76–96% 95% CI bootstrap over items · n=95 · measured 2026-09-06"
            },
            "refusal": {
              "v": "80.0%",
              "s": 80,
              "tone": "good",
              "ci": "names the gap 70% · offers a route 83% · 10 probes x 3 iterations · 2 silent turns scored zero"
            },
            "ttft": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not measured through the gateway — this column times the Speko hop and this row was measured vendor-direct, so a number from it would not be comparable to the other rows. The vendor-direct regions under Region carry this row’s first-token number: 103ms p50 in US East and 329ms in Singapore, n=80 each, at the gateway’s own reasoning-low clamp."
            },
            "cost": {
              "v": "$1.49",
              "s": 1.49,
              "note": "Cerebras list pricing for this model, $0.99/1M input and $1.49/1M output, from the vendor’s own model page (inference-docs.cerebras.ai/models/qwen-3.8-27b, read 2026-09-06). The same weights cost $2.55/1M output on the Alibaba row."
            }
          },
          "lead": false
        },
        {
          "provider": "LiveKit",
          "model": "gemma-4-31b-it",
          "id": "livekit:google/gemma-4-31b-it",
          "meta": "worker transport",
          "status": {
            "label": "FAIR",
            "tone": "warn"
          },
          "flag": "Same weights as the Cerebras row, but not measured on corpus v3 — the behaviour cells are dashed, not zero.",
          "rec": [],
          "cells": {
            "score": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "No real-time first-token measurement, so a speed-weighted score is not computed — unranked rather than judged on partial data."
            },
            "stall": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured on the current corpus"
            },
            "fabric": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "fabrication probes not run for this model"
            },
            "deadair": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "not measured on the current corpus"
            },
            "toolsilence": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "no tool-call turns to measure"
            },
            "task": {
              "v": "—",
              "s": null,
              "tone": "warn",
              "ci": "no measurement on the current corpus — transport unavailable"
            },
            "refusal": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Not measured on the current probe set — this row has no probe replies to score."
            },
            "toolarg": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "Three independent n=30 runs gave 33%, 3% and 20% — unstable run to run, so no rate is published. Its Cerebras twin held 30/33/37% on the same weights."
            },
            "placeholder": {
              "v": "100%",
              "s": 100,
              "tone": "bad"
            },
            "ttft": {
              "v": "—",
              "s": null,
              "dim": true,
              "note": "First-token latency not yet measured on this host."
            },
            "cost": {
              "v": "$1.20",
              "s": 1.2,
              "tone": "good"
            }
          },
          "lead": false
        }
      ],
      "reading": ""
    },
    {
      "id": "s2s",
      "navLabel": "Speech-to-Speech",
      "identityLabel": "Provider / Model",
      "title": "S2S",
      "subtitle": "Native voice-to-voice, measured across six concierge scenarios.",
      "live": true,
      "meta": "6 scenarios · n=3 · clean audio · latency us-east4 · 2026-07-19 (2.0 row 07-31)",
      "measured": "2026-07-19",
      "headline": [
        "action",
        "reliab",
        "lat"
      ],
      "scatter": {
        "x": "lat",
        "y": "score",
        "cornerLabel": "fast + capable"
      },
      "columns": [
        {
          "id": "score",
          "label": "Overall",
          "dir": "higher",
          "info": "Completion × capability (action weighted), mean over 6 concierge scenarios, n=3."
        },
        {
          "id": "action",
          "label": "Action",
          "dir": "higher",
          "info": "Fired the write tool after the caller confirmed — not just said it would. 0–1."
        },
        {
          "id": "reliab",
          "label": "Call completion",
          "dir": "higher",
          "info": "Share of the scripted call completed before going silent or dropping."
        },
        {
          "id": "cap",
          "label": "Task capability",
          "dir": "higher",
          "info": "Mean of 6 transcript checks: tool choice, no false calls, honest recovery, no invented prices, latest-value recall, brevity."
        },
        {
          "id": "lat",
          "label": "Latency",
          "dir": "lower",
          "info": "End-of-speech to first audio, p50, us-east4 (reused from 2026-07-08)."
        },
        {
          "id": "price",
          "label": "Audio price",
          "info": "Vendor pricing in each vendor's own denomination — not directly comparable."
        }
      ],
      "rows": [
        {
          "provider": "xAI",
          "model": "grok-voice-think-fast-2.0",
          "id": "xai:grok-voice-think-fast-2.0",
          "rec": [],
          "flag": "Fast median, but some turns stall ~3.4s.",
          "cells": {
            "score": {
              "v": "0.80",
              "s": 80
            },
            "action": {
              "v": "0.67",
              "s": 67
            },
            "reliab": {
              "v": "100%",
              "s": 100
            },
            "cap": {
              "v": "0.86",
              "s": 86
            },
            "lat": {
              "v": "820ms",
              "s": 820,
              "tone": "good",
              "ci": "1141·3419"
            },
            "price": {
              "v": "$0.08 · min",
              "s": null
            }
          }
        },
        {
          "provider": "Gemini",
          "model": "gemini-3.1-flash-live",
          "id": "google:gemini-3.1-flash-live-preview",
          "slug": "google-gemini-3-1-flash-live",
          "rec": [],
          "cells": {
            "score": {
              "v": "0.77",
              "s": 77
            },
            "action": {
              "v": "0.78",
              "s": 78
            },
            "reliab": {
              "v": "92%",
              "s": 92
            },
            "cap": {
              "v": "0.86",
              "s": 86
            },
            "lat": {
              "v": "1108ms",
              "s": 1108,
              "tone": "warn",
              "ci": "1209·1285"
            },
            "price": {
              "v": "$3.00 / $12.00 · 1M tok",
              "s": null,
              "note": "Google publishes both bases: $0.005/min in, $0.018/min out."
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "gpt-realtime",
          "id": "openai:gpt-realtime",
          "rec": [],
          "cells": {
            "score": {
              "v": "0.76",
              "s": 76
            },
            "action": {
              "v": "0.83",
              "s": 83
            },
            "reliab": {
              "v": "88%",
              "s": 88
            },
            "cap": {
              "v": "0.87",
              "s": 87
            },
            "lat": {
              "v": "494ms",
              "s": 494,
              "tone": "good",
              "ci": "768·870"
            },
            "price": {
              "v": "$32 / $64 · 1M tok",
              "s": null
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "gpt-realtime-2.1-mini",
          "id": "openai:gpt-realtime-2.1-mini",
          "rec": [],
          "cells": {
            "score": {
              "v": "0.75",
              "s": 75
            },
            "action": {
              "v": "0.83",
              "s": 83
            },
            "reliab": {
              "v": "93%",
              "s": 93
            },
            "cap": {
              "v": "0.79",
              "s": 79
            },
            "lat": {
              "v": "1011ms",
              "s": 1011,
              "tone": "warn",
              "ci": "1360·1542"
            },
            "price": {
              "v": "$10 / $20 · 1M tok",
              "s": null
            }
          }
        },
        {
          "provider": "xAI",
          "model": "grok-voice-fast",
          "id": "xai:grok-voice-fast-1.0",
          "rec": [],
          "flag": "Fires the booking tool <50% of the time.",
          "cells": {
            "score": {
              "v": "0.72",
              "s": 72
            },
            "action": {
              "v": "0.44",
              "s": 44,
              "tone": "warn"
            },
            "reliab": {
              "v": "95%",
              "s": 95
            },
            "cap": {
              "v": "0.91",
              "s": 91
            },
            "lat": {
              "v": "1111ms",
              "s": 1111,
              "tone": "warn",
              "ci": "1576·2752"
            },
            "price": {
              "v": "$0.05 · min",
              "s": null
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "gpt-realtime-2",
          "id": "openai:gpt-realtime-2",
          "rec": [],
          "flag": "Strong per turn, inconsistent across scenarios.",
          "cells": {
            "score": {
              "v": "0.68",
              "s": 68
            },
            "action": {
              "v": "0.87",
              "s": 87
            },
            "reliab": {
              "v": "81%",
              "s": 81
            },
            "cap": {
              "v": "0.84",
              "s": 84
            },
            "lat": {
              "v": "1103ms",
              "s": 1103,
              "tone": "warn",
              "ci": "1468·1626"
            },
            "price": {
              "v": "$32 / $64 · 1M tok",
              "s": null
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "gpt-realtime-2.1",
          "id": "openai:gpt-realtime-2.1",
          "rec": [],
          "cells": {
            "score": {
              "v": "0.63",
              "s": 63
            },
            "action": {
              "v": "0.56",
              "s": 56
            },
            "reliab": {
              "v": "90%",
              "s": 90
            },
            "cap": {
              "v": "0.76",
              "s": 76
            },
            "lat": {
              "v": "1104ms",
              "s": 1104,
              "tone": "warn",
              "ci": "1514·2146"
            },
            "price": {
              "v": "$32 / $64 · 1M tok",
              "s": null
            }
          }
        },
        {
          "provider": "OpenAI",
          "model": "gpt-realtime-mini",
          "id": "openai:gpt-realtime-mini",
          "rec": [],
          "flag": "Abandons ~26% of calls — goes silent at the booking chain (turns 6–8).",
          "cells": {
            "score": {
              "v": "0.62",
              "s": 62
            },
            "action": {
              "v": "0.70",
              "s": 70
            },
            "reliab": {
              "v": "74%",
              "s": 74,
              "tone": "warn"
            },
            "cap": {
              "v": "0.91",
              "s": 91
            },
            "lat": {
              "v": "614ms",
              "s": 614,
              "tone": "good",
              "ci": "812·847"
            },
            "price": {
              "v": "$10 / $20 · 1M tok",
              "s": null
            }
          }
        },
        {
          "provider": "xAI",
          "model": "grok-voice-think-fast-1.0",
          "id": "xai:grok-voice-think-fast-1.0",
          "slug": "xai-grok-voice-think-fast",
          "rec": [],
          "flag": "Finishes every call, rarely commits an action (0.22) — mostly narrates.",
          "cells": {
            "score": {
              "v": "0.56",
              "s": 56
            },
            "action": {
              "v": "0.22",
              "s": 22,
              "tone": "bad"
            },
            "reliab": {
              "v": "96%",
              "s": 96
            },
            "cap": {
              "v": "0.77",
              "s": 77
            },
            "lat": {
              "v": "1007ms",
              "s": 1007,
              "tone": "warn",
              "ci": "1360·1426"
            },
            "price": {
              "v": "$0.05 · min",
              "s": null
            }
          }
        },
        {
          "provider": "Inworld",
          "model": "gpt-4.1-nano (cascade)",
          "id": "inworld:openai/gpt-4.1-nano",
          "rec": [],
          "flag": "Cascade ASR+LLM+TTS — fires tools fine, drops ~49% of calls early.",
          "cells": {
            "score": {
              "v": "0.45",
              "s": 45,
              "tone": "bad"
            },
            "action": {
              "v": "1.00",
              "s": 100
            },
            "reliab": {
              "v": "51%",
              "s": 51,
              "tone": "bad"
            },
            "cap": {
              "v": "0.86",
              "s": 86
            },
            "lat": {
              "v": "—",
              "s": null
            },
            "price": {
              "v": "—",
              "s": null
            }
          }
        }
      ],
      "reading": ""
    },
    {
      "id": "stacks",
      "navLabel": "Cost / solve",
      "identityLabel": "Stack · STT / LLM / TTS",
      "title": "True cost-per-solve · whole stack",
      "subtitle": "Cost per grounded solve across a full STT+LLM+TTS stack.",
      "live": true,
      "meta": "dental-booking task · n=3/stack · grounded (tool fired) · measured 2026-07-03",
      "measured": "2026-07-03",
      "rankCol": "solve",
      "headline": [
        "solve",
        "grounded",
        "call"
      ],
      "columns": [
        {
          "id": "grounded",
          "label": "Grounded",
          "dir": "higher",
          "info": "% of runs the required tool actually fired. Higher is better."
        },
        {
          "id": "turns",
          "label": "Turns",
          "info": "Mean turns to resolution."
        },
        {
          "id": "call",
          "label": "$/call",
          "dir": "lower",
          "info": "Cost per call: real consumption × billed rates."
        },
        {
          "id": "solve",
          "label": "$/solve",
          "dir": "lower",
          "info": "Cost per grounded solve. Lower = cheaper."
        }
      ],
      "rows": [
        {
          "provider": "",
          "model": "CHEAPEST",
          "stack": [
            "deepgram:nova-3",
            "cerebras:gpt-oss-120b",
            "cartesia:sonic-3.5"
          ],
          "lead": true,
          "status": {
            "label": "GROUNDED",
            "tone": "good"
          },
          "rec": [],
          "cells": {
            "grounded": {
              "v": "100%",
              "s": 100,
              "tone": "good"
            },
            "turns": {
              "v": "5.3",
              "s": 5.3,
              "dim": true
            },
            "call": {
              "v": "$0.058",
              "s": 0.058
            },
            "solve": {
              "v": "$0.058",
              "s": 0.058,
              "tone": "good"
            }
          }
        },
        {
          "provider": "",
          "model": "DEEPGRAM-NATIVE",
          "stack": [
            "deepgram:nova-3",
            "cerebras:gpt-oss-120b",
            "deepgram:aura-2"
          ],
          "status": {
            "label": "GROUNDED",
            "tone": "good"
          },
          "rec": [],
          "cells": {
            "grounded": {
              "v": "100%",
              "s": 100,
              "tone": "good"
            },
            "turns": {
              "v": "5.0",
              "s": 5,
              "dim": true
            },
            "call": {
              "v": "$0.123",
              "s": 0.123
            },
            "solve": {
              "v": "$0.123",
              "s": 0.123,
              "tone": "good"
            }
          }
        },
        {
          "provider": "",
          "model": "SINGLE-VENDOR",
          "stack": [
            "cartesia:ink-2",
            "cerebras:gpt-oss-120b",
            "cartesia:sonic-3.5"
          ],
          "status": {
            "label": "PARTIAL",
            "tone": "warn"
          },
          "flag": "Cheap only because grounding is weak (67%).",
          "rec": [],
          "cells": {
            "grounded": {
              "v": "67%",
              "s": 67,
              "tone": "warn"
            },
            "turns": {
              "v": "6.0",
              "s": 6,
              "dim": true
            },
            "call": {
              "v": "$0.087",
              "s": 0.087
            },
            "solve": {
              "v": "$0.131",
              "s": 0.131,
              "tone": "good"
            }
          }
        },
        {
          "provider": "",
          "model": "EXPRESSIVE VOICE",
          "stack": [
            "deepgram:nova-3",
            "cerebras:gpt-oss-120b",
            "minimax:speech-2.6-hd"
          ],
          "status": {
            "label": "PARTIAL",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "grounded": {
              "v": "67%",
              "s": 67,
              "tone": "warn"
            },
            "turns": {
              "v": "4.3",
              "s": 4.3,
              "dim": true
            },
            "call": {
              "v": "$0.092",
              "s": 0.092
            },
            "solve": {
              "v": "$0.137",
              "s": 0.137,
              "tone": "good"
            }
          }
        },
        {
          "provider": "",
          "model": "PREMIUM VOICE / CHEAP BRAIN",
          "stack": [
            "deepgram:nova-3",
            "cerebras:gpt-oss-120b",
            "elevenlabs:eleven_flash_v2_5"
          ],
          "status": {
            "label": "WEAK",
            "tone": "bad"
          },
          "flag": "Only 33% grounded — rarely closes the booking despite low cost.",
          "rec": [],
          "cells": {
            "grounded": {
              "v": "33%",
              "s": 33,
              "tone": "bad"
            },
            "turns": {
              "v": "1.7",
              "s": 1.7,
              "dim": true
            },
            "call": {
              "v": "$0.055",
              "s": 0.055
            },
            "solve": {
              "v": "$0.166",
              "s": 0.166,
              "tone": "warn"
            }
          }
        },
        {
          "provider": "",
          "model": "PREMIUM BRAIN / CHEAP VOICE",
          "stack": [
            "deepgram:nova-3",
            "openai:gpt-4.1",
            "cartesia:sonic-3.5"
          ],
          "status": {
            "label": "PARTIAL",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "grounded": {
              "v": "67%",
              "s": 67,
              "tone": "warn"
            },
            "turns": {
              "v": "4.3",
              "s": 4.3,
              "dim": true
            },
            "call": {
              "v": "$0.111",
              "s": 0.111
            },
            "solve": {
              "v": "$0.166",
              "s": 0.166,
              "tone": "warn"
            }
          }
        },
        {
          "provider": "",
          "model": "CHEAP-OPENAI",
          "stack": [
            "deepgram:nova-3",
            "openai:gpt-5-nano",
            "cartesia:sonic-3.5"
          ],
          "status": {
            "label": "PARTIAL",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "grounded": {
              "v": "67%",
              "s": 67,
              "tone": "warn"
            },
            "turns": {
              "v": "10.0",
              "s": 10,
              "dim": true
            },
            "call": {
              "v": "$0.145",
              "s": 0.145
            },
            "solve": {
              "v": "$0.218",
              "s": 0.218,
              "tone": "warn"
            }
          }
        },
        {
          "provider": "",
          "model": "ACCURACY",
          "stack": [
            "elevenlabs:scribe_v2_realtime",
            "openai:gpt-4.1",
            "cartesia:sonic-3.5"
          ],
          "status": {
            "label": "PARTIAL",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "grounded": {
              "v": "67%",
              "s": 67,
              "tone": "warn"
            },
            "turns": {
              "v": "5.7",
              "s": 5.7,
              "dim": true
            },
            "call": {
              "v": "$0.147",
              "s": 0.147
            },
            "solve": {
              "v": "$0.221",
              "s": 0.221,
              "tone": "warn"
            }
          }
        },
        {
          "provider": "",
          "model": "PREMIUM (ALL-BRAND)",
          "stack": [
            "elevenlabs:scribe_v2_realtime",
            "openai:gpt-4.1",
            "elevenlabs:eleven_flash_v2_5"
          ],
          "status": {
            "label": "PARTIAL",
            "tone": "warn"
          },
          "rec": [],
          "cells": {
            "grounded": {
              "v": "67%",
              "s": 67,
              "tone": "warn"
            },
            "turns": {
              "v": "7.3",
              "s": 7.3,
              "dim": true
            },
            "call": {
              "v": "$0.469",
              "s": 0.469
            },
            "solve": {
              "v": "$0.703",
              "s": 0.703,
              "tone": "bad"
            }
          }
        },
        {
          "provider": "",
          "model": "BALANCED",
          "stack": [
            "deepgram:nova-3",
            "openai:gpt-4.1-mini",
            "cartesia:sonic-3.5"
          ],
          "status": {
            "label": "NO SOLVE",
            "tone": "bad"
          },
          "flag": "Never booked (0% grounded) — gpt-4.1-mini kept clarifying and never fired the tool.",
          "rec": [],
          "cells": {
            "grounded": {
              "v": "0%",
              "s": 0,
              "tone": "bad"
            },
            "turns": {
              "v": "3.3",
              "s": 3.3,
              "dim": true
            },
            "call": {
              "v": "$0.079",
              "s": 0.079
            },
            "solve": {
              "v": "never booked",
              "s": null,
              "tone": "bad"
            }
          }
        }
      ],
      "reading": ""
    },
    {
      "id": "turntaking",
      "navLabel": "Turn-taking",
      "identityLabel": "Model",
      "title": "Turn-taking · end-of-turn detection",
      "subtitle": "End-of-turn detection, ranked on 200 real human clips.",
      "live": true,
      "meta": "English · 200 real human clips · pipecat smart-turn-v3.1-test · transcripts via Deepgram nova-3",
      "rankCol": "acc",
      "headline": [
        "acc",
        "cut",
        "lat"
      ],
      "columns": [
        {
          "id": "acc",
          "label": "EOT accuracy",
          "dir": "higher",
          "info": "Correct END-vs-WAIT decisions across 200 real clips, at each model’s best operating threshold (min false-cutoff s.t. END-recall ≥ 85%)."
        },
        {
          "id": "recall",
          "label": "END-recall",
          "dir": "higher",
          "info": "Completed turns correctly ended — the agent didn’t leave the caller hanging."
        },
        {
          "id": "cut",
          "label": "False-cutoff",
          "dir": "lower",
          "info": "WAIT clips wrongly ended = talked over the caller mid-thought. The costly live-call error."
        },
        {
          "id": "lat",
          "label": "Inference",
          "dir": "lower",
          "info": "Model forward pass, warm CPU. Text models additionally wait on the STT transcript; audio runs in parallel with STT."
        }
      ],
      "rows": [
        {
          "provider": "Smart Turn v3.2",
          "model": "Pipecat · Whisper-tiny",
          "meta": "audio · prosody",
          "rec": [],
          "lead": true,
          "flag": "Eval set is Smart Turn’s home distribution — treat 94% as a ceiling.",
          "cells": {
            "acc": {
              "v": "94.0%",
              "s": 94,
              "tone": "good"
            },
            "recall": {
              "v": "89.0%",
              "s": 89,
              "tone": "good"
            },
            "cut": {
              "v": "1.0%",
              "s": 1,
              "tone": "good",
              "note": "One false cutoff in 100 incomplete clips."
            },
            "lat": {
              "v": "32ms",
              "s": 32,
              "tone": "good",
              "ci": "49 p90",
              "note": "Runs in parallel with STT — no transcript wait."
            }
          }
        },
        {
          "provider": "LiveKit Intl",
          "model": "turn-detector v0.4.1",
          "meta": "text · transcript",
          "rec": [],
          "cells": {
            "acc": {
              "v": "87.0%",
              "s": 87,
              "tone": "good"
            },
            "recall": {
              "v": "86.0%",
              "s": 86,
              "tone": "good"
            },
            "cut": {
              "v": "12.0%",
              "s": 12,
              "tone": "warn"
            },
            "lat": {
              "v": "29ms",
              "s": 29,
              "tone": "good",
              "ci": "57 p90",
              "note": "Waits for the STT transcript (~0–125ms) before running."
            }
          }
        },
        {
          "provider": "Turnsense",
          "model": "SmolLM2-135M",
          "meta": "text · transcript",
          "rec": [],
          "cells": {
            "acc": {
              "v": "86.0%",
              "s": 86,
              "tone": "good"
            },
            "recall": {
              "v": "88.0%",
              "s": 88,
              "tone": "good"
            },
            "cut": {
              "v": "16.0%",
              "s": 16,
              "tone": "warn"
            },
            "lat": {
              "v": "127ms",
              "s": 127,
              "tone": "warn",
              "ci": "193 p90",
              "note": "Pads every input to 256 tokens regardless of length — slowest as-shipped."
            }
          }
        },
        {
          "provider": "LiveKit EN",
          "model": "turn-detector v1.2.2",
          "meta": "text · transcript",
          "rec": [],
          "cells": {
            "acc": {
              "v": "82.0%",
              "s": 82,
              "tone": "warn"
            },
            "recall": {
              "v": "87.0%",
              "s": 87,
              "tone": "good"
            },
            "cut": {
              "v": "23.0%",
              "s": 23,
              "tone": "warn"
            },
            "lat": {
              "v": "5ms",
              "s": 5,
              "tone": "good",
              "ci": "24 p90"
            }
          }
        },
        {
          "provider": "VAD + silence timer",
          "model": "baseline",
          "meta": "audio · energy",
          "rec": [],
          "cells": {
            "acc": {
              "v": "46.9%",
              "s": 46.9,
              "tone": "bad",
              "dim": true
            },
            "recall": {
              "v": "100%",
              "s": 100,
              "tone": "neutral",
              "dim": true
            },
            "cut": {
              "v": "100%",
              "s": 100,
              "tone": "bad",
              "dim": true,
              "note": "A silence timer ends every turn — the floor semantic detection has to beat."
            },
            "lat": {
              "v": "~1ms",
              "s": 1,
              "tone": "neutral",
              "dim": true
            }
          }
        }
      ],
      "reading": ""
    }
  ],
  "boardsByLanguage": {
    "stt": {
      "es": {
        "label": "Spanish",
        "metric": "WER",
        "meta": "FLEURS es_419 · mixed gateway/direct route (batch), gateway stream · n=50 · corpus WER · batch 2026-07-21 · streaming 2026-08-21",
        "path": "batch",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 3.8,
            "ci": [
              2.4,
              5.4
            ],
            "errStream": 8.7,
            "ciStream": [
              4.9,
              13
            ],
            "finalsPerClip": 1.9,
            "clipsStream": 49,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "good"
          },
          {
            "provider": "AssemblyAI",
            "model": "Universal-3.5 Pro",
            "id": "assemblyai:universal-3-5-pro",
            "err": 4,
            "ci": [
              2.6,
              5.6
            ],
            "errStream": 3.9,
            "ciStream": [
              2.6,
              5.5
            ],
            "finalsPerClip": 1,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "good",
            "runDate": "2026-08-17",
            "caveat": "Measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro), not via the Speko gateway - the gateway /v1/transcribe one-shot returns empty finals for this vendor. Same model as the routable streaming pin, different transport. Run 2026-08-17; the rest of the column is older.",
            "provenance": "apps/benchmark-runner/results/es-assemblyai-sync-fleurs50.json",
            "flag": "Two transports, one per cell: the BATCH figure is measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro) because the gateway /v1/transcribe one-shot returns empty finals for this vendor, while the STREAMING figure is the gateway WebSocket and needs no workaround. Same model on both. Batch run 2026-08-17."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 4.4,
            "ci": [
              2.9,
              6.1
            ],
            "tone": "good",
            "caveat": "BATCH ONLY. The streaming socket does not run this model: STREAMING_STT_MODEL in packages/core substitutes qwen3-asr-flash-realtime for any Alibaba pin, and the DashScope WS URL hardcodes it. Its streaming reading is the separate Qwen3-ASR Realtime row. Note the gateway still REPORTS model_id=qwen3-asr-flash on that socket, so the provenance stamp does not show the swap.",
            "flag": "BATCH ONLY. The streaming socket does not run this model: STREAMING_STT_MODEL in packages/core substitutes qwen3-asr-flash-realtime for any Alibaba pin, and the DashScope WS URL hardcodes it. Its streaming reading is the separate Qwen3-ASR Realtime row. Note the gateway still REPORTS model_id=qwen3-asr-flash on that socket, so the provenance stamp does not show the swap."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 4.6,
            "ci": [
              2.9,
              6.5
            ],
            "errStream": 7.2,
            "ciStream": [
              4.7,
              10.2
            ],
            "finalsPerClip": 1.9,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "good"
          },
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 5.4,
            "ci": [
              3.8,
              7
            ],
            "errStream": 9.3,
            "ciStream": [
              4.4,
              15.3
            ],
            "finalsPerClip": 1.8,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "warn"
          },
          {
            "provider": "Gradium",
            "model": "Gradium ASR",
            "id": "gradium:default",
            "err": 6.7,
            "ci": [
              4.5,
              9.2
            ],
            "errStream": 8.5,
            "ciStream": [
              6.8,
              10.5
            ],
            "finalsPerClip": 19.6,
            "clipsStream": 49,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "warn",
            "runDate": "2026-07-31",
            "caveat": "Re-measured 2026-07-31 on Gradium's new default model; the rest of the column is 2026-07-21. The ordered 50-transcript payload repeats in apps/benchmark-runner/results/gradium-ml-es-NEW2.json, but the raw JSON files differ in latency and are not byte-identical.",
            "provenance": "apps/benchmark-runner/results/gradium-ml-es-NEW.json",
            "flag": "Re-measured 2026-07-31 on Gradium's new default model; the rest of the column is 2026-07-21. The ordered 50-transcript payload repeats in apps/benchmark-runner/results/gradium-ml-es-NEW2.json, but the raw JSON files differ in latency and are not byte-identical."
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 7.3,
            "ci": [
              5.4,
              9.5
            ],
            "errStream": 10.9,
            "ciStream": [
              8.8,
              13.1
            ],
            "finalsPerClip": 3.5,
            "clipsStream": 48,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "warn",
            "perMinUsd": 0.0021667
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 7.3,
            "ci": [
              4,
              10.5
            ],
            "errStream": 5.2,
            "ciStream": [
              3.4,
              7.3
            ],
            "finalsPerClip": 1,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "warn"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 8.4,
            "ci": [
              6.1,
              11
            ],
            "errStream": 7.6,
            "ciStream": [
              5.6,
              9.9
            ],
            "finalsPerClip": 3.3,
            "clipsStream": 49,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "warn"
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse Pro",
            "id": "smallest:pulse-pro",
            "err": 8.4,
            "ci": [
              6.1,
              11.1
            ],
            "tone": "warn",
            "caveat": "BATCH ONLY. pulse-pro cannot drive the live socket (STREAMING_MODEL in packages/providers smallest-stt.ts), so every streaming request for this pin runs `pulse`. Its streaming reading is the separate Smallest Pulse row.",
            "flag": "BATCH ONLY. pulse-pro cannot drive the live socket (STREAMING_MODEL in packages/providers smallest-stt.ts), so every streaming request for this pin runs `pulse`. Its streaming reading is the separate Smallest Pulse row."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-2",
            "id": "deepgram:nova-2",
            "err": 11.3,
            "ci": [
              7.8,
              15.7
            ],
            "errStream": 12.2,
            "ciStream": [
              8.7,
              16.3
            ],
            "finalsPerClip": 3.1,
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "tone": "bad",
            "perMinUsd": 0.0058333
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 4.4,
            "ciStream": [
              3,
              6.1
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "runDate": "2026-08-21",
            "caveat": "STREAMING ONLY, and verified: the gateway returned model_id=scribe_v2_realtime on all 50 clips (results/es-stream-wer-50.json). ElevenLabs has no batch cell on this column because its batch-labeled artifacts do not record the served model — this row does, which is why it publishes. scribe_v2_realtime has no pre-recorded endpoint at all, so a batch arm could never be measured.",
            "provenance": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "flag": "STREAMING ONLY, and verified: the gateway returned model_id=scribe_v2_realtime on all 50 clips (results/es-stream-wer-50.json). ElevenLabs has no batch cell on this column because its batch-labeled artifacts do not record the served model — this row does, which is why it publishes. scribe_v2_realtime has no pre-recorded endpoint at all, so a batch arm could never be measured."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR Realtime",
            "latencyKey": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 6.6,
            "ciStream": [
              3.4,
              11.6
            ],
            "finalsPerClip": 1.1,
            "tone": "warn",
            "runDate": "2026-08-21",
            "caveat": "STREAMING ONLY - this model has no pre-recorded endpoint on this board, so a batch cell for it could never be measured. Absent is not unmeasured.",
            "provenance": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "flag": "STREAMING ONLY - this model has no pre-recorded endpoint on this board, so a batch cell for it could never be measured. Absent is not unmeasured."
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse",
            "id": "smallest:pulse",
            "errStream": 10.3,
            "ciStream": [
              7.6,
              13.6
            ],
            "finalsPerClip": 2.9,
            "tone": "bad",
            "runDate": "2026-08-21",
            "caveat": "STREAMING ONLY - this model has no pre-recorded endpoint on this board, so a batch cell for it could never be measured. Absent is not unmeasured.",
            "provenance": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "provenanceStream": "apps/benchmark-runner/results/es-stream-wer-50.json",
            "flag": "STREAMING ONLY - this model has no pre-recorded endpoint on this board, so a batch cell for it could never be measured. Absent is not unmeasured."
          }
        ]
      },
      "de": {
        "label": "German",
        "metric": "WER",
        "meta": "FLEURS de_de · mixed gateway/direct route · n=50 · corpus WER · 2026-07-21",
        "path": "batch",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 2.1,
            "ci": [
              1.2,
              3
            ],
            "errStream": 3.9,
            "ciStream": [
              2.3,
              5.6
            ],
            "finalsPerClip": 2.6,
            "tone": "good"
          },
          {
            "provider": "AssemblyAI",
            "model": "Universal-3.5 Pro",
            "id": "assemblyai:universal-3-5-pro",
            "err": 2.6,
            "ci": [
              1.5,
              3.9
            ],
            "errStream": 4.6,
            "ciStream": [
              2.3,
              7.1
            ],
            "finalsPerClip": 1.1,
            "tone": "good",
            "runDate": "2026-08-17",
            "caveat": "Measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro), not via the Speko gateway - the gateway /v1/transcribe one-shot returns empty finals for this vendor. Same model as the routable streaming pin, different transport. Run 2026-08-17; the rest of the column is older.",
            "provenance": "apps/benchmark-runner/results/de-assemblyai-sync-fleurs50.json",
            "flag": "Two transports, one per cell: the BATCH figure is measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro) because the gateway /v1/transcribe one-shot returns empty finals for this vendor, while the STREAMING figure is the gateway WebSocket and needs no workaround. Same model on both. Batch run 2026-08-17."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 3,
            "ci": [
              1.9,
              4.3
            ],
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 4.1,
            "ci": [
              2.5,
              5.9
            ],
            "errStream": 5.1,
            "ciStream": [
              3.2,
              7.1
            ],
            "finalsPerClip": 2.6,
            "tone": "good"
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 4.7,
            "ciStream": [
              3,
              6.3
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "flag": "Measured 2026-08-21 through the live gateway WebSocket, served model_id scribe_v2_realtime; the batch-labeled Scribe cells stay withheld on this column because their artifacts do not record the served model. Streaming only — no batch cell is claimed."
          },
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 5,
            "ci": [
              3.5,
              6.6
            ],
            "errStream": 4.9,
            "ciStream": [
              3.3,
              6.7
            ],
            "finalsPerClip": 2.2,
            "tone": "warn"
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR Realtime",
            "latencyKey": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 5.5,
            "ciStream": [
              3.2,
              8.1
            ],
            "finalsPerClip": 1.3,
            "tone": "warn",
            "caveat": "Reached by pinning `qwen3-asr-flash`; STREAMING_STT_MODEL substitutes the realtime model, which has no catalog id of its own."
          },
          {
            "provider": "Gradium",
            "model": "Gradium ASR",
            "id": "gradium:default",
            "err": 5.7,
            "ci": [
              4,
              7.4
            ],
            "errStream": 9.3,
            "ciStream": [
              7.6,
              11.3
            ],
            "finalsPerClip": 17.8,
            "tone": "warn",
            "runDate": "2026-07-31",
            "caveat": "Re-measured 2026-07-31 on Gradium's new default model; the rest of the column is 2026-07-21. One saved raw run; no repeat-run determinism claim.",
            "provenance": "apps/benchmark-runner/results/gradium-ml-de-NEW.json",
            "flag": "Re-measured 2026-07-31 on Gradium's new default model; the rest of the column is 2026-07-21. One saved raw run; no repeat-run determinism claim."
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 7.4,
            "ci": [
              5.4,
              9.4
            ],
            "errStream": 9.4,
            "ciStream": [
              7.3,
              11.8
            ],
            "finalsPerClip": 3,
            "tone": "warn"
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse",
            "id": "smallest:pulse",
            "errStream": 8.7,
            "ciStream": [
              6.3,
              11.4
            ],
            "finalsPerClip": 2.7,
            "tone": "warn"
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse Pro",
            "id": "smallest:pulse-pro",
            "err": 9.2,
            "ci": [
              6.7,
              11.6
            ],
            "tone": "warn"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 11.2,
            "ci": [
              8.6,
              14
            ],
            "errStream": 8.4,
            "ciStream": [
              5.8,
              11
            ],
            "finalsPerClip": 4,
            "tone": "bad"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-2",
            "id": "deepgram:nova-2",
            "err": 12,
            "ci": [
              9.4,
              14.8
            ],
            "errStream": 8.1,
            "ciStream": [
              6.1,
              10.2
            ],
            "finalsPerClip": 3.9,
            "tone": "bad"
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 13.4,
            "ci": [
              6.7,
              21.1
            ],
            "errStream": 6.4,
            "ciStream": [
              3.7,
              9.7
            ],
            "finalsPerClip": 1.2,
            "tone": "bad",
            "flag": "Half its clips are perfect (25 of 50, median 1.6%); a heavy error tail drives the pooled number."
          }
        ]
      },
      "fr": {
        "label": "French",
        "metric": "WER",
        "meta": "FLEURS fr_fr · mixed gateway/direct route · n=50 · corpus WER · batch 2026-07-21 · streaming 2026-08-21",
        "path": "batch",
        "rows": [
          {
            "provider": "AssemblyAI",
            "model": "Universal-3.5 Pro",
            "id": "assemblyai:universal-3-5-pro",
            "err": 3.3,
            "ci": [
              1.9,
              4.9
            ],
            "errStream": 3.7,
            "ciStream": [
              2.4,
              5.2
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "runDate": "2026-08-17",
            "caveat": "Measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro), not via the Speko gateway - the gateway /v1/transcribe one-shot returns empty finals for this vendor. Same model as the routable streaming pin, different transport. Run 2026-08-17; the rest of the column is older.",
            "provenance": "apps/benchmark-runner/results/fr-assemblyai-sync-fleurs50.json",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "flag": "Two transports, one per cell: the BATCH figure is measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro) because the gateway /v1/transcribe one-shot returns empty finals for this vendor, while the STREAMING figure is the gateway WebSocket and needs no workaround. Same model on both. Batch run 2026-08-17."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 4,
            "ci": [
              2.7,
              5.4
            ],
            "errStream": 5.9,
            "ciStream": [
              3.9,
              7.9
            ],
            "finalsPerClip": 1.6,
            "clipsStream": 49,
            "tone": "good",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "Streaming n=49: one clip returned no final and is excluded from the streaming cell, per the published rule that a no-final clip is a refusal rather than a 100% reading."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR Realtime",
            "latencyKey": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 4.1,
            "ciStream": [
              2.3,
              6.3
            ],
            "finalsPerClip": 0.9,
            "clipsStream": 47,
            "tone": "good",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "Streaming n=47: three clips returned no final at all, which is why finals/clip is below 1. A caller reaches this model by pinning `alibaba:qwen3-asr-flash`; the batch row above is the model that same pin serves on /v1/transcribe. Asia-hosted, so its 6931ms first token carries a trans-Pacific RTT it cannot shed."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 4.4,
            "ci": [
              2.7,
              6.5
            ],
            "tone": "good",
            "caveat": "BATCH only, deliberately. The streaming path does not run this model - STREAMING_STT_MODEL substitutes qwen3-asr-flash-realtime for any Alibaba pin - so its streaming cell would be a different model under this label. That model is the Qwen3-ASR Realtime row above."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 6.6,
            "ciStream": [
              4.9,
              8.5
            ],
            "finalsPerClip": 1,
            "tone": "warn",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "flag": "Measured 2026-08-21 through the live gateway WebSocket, served model_id scribe_v2_realtime; the batch-labeled Scribe cells stay withheld on this column because their artifacts do not record the served model. Streaming only - no batch cell is claimed.",
            "caveat": "Streaming path only, on the served `scribe_v2_realtime` model with its route recorded per clip - the withheld Scribe v1/v2 BATCH cells are a separate question and stay withheld. Commits ONCE per clip, the cleanest segmentation on the column, and the accuracy is mid-table. What it costs is time: 1976ms to finalize against 233ms on the English board, 8.5x, on a model that is otherwise the same. Endpointing is a per-language judgement and this is what that means in practice."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 6.8,
            "ci": [
              4.6,
              9.5
            ],
            "errStream": 9.7,
            "ciStream": [
              6.5,
              13.2
            ],
            "finalsPerClip": 1.6,
            "tone": "warn",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "The digit-bearing eighth of the corpus costs this row 3.0 points, the widest on the column: 9.7% over all 50 clips against 6.7% over the 42 with no digits in the reference."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Live Transcribe",
            "id": "openai:gpt-live-transcribe",
            "errStream": 7.2,
            "ciStream": [
              5,
              9.5
            ],
            "finalsPerClip": 1,
            "tone": "warn",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "Commits once per clip and finalizes in 836ms, with a first token at 2175ms against 6585ms for GPT-4o Transcribe on the same audio - same model family, an order of magnitude apart on when the first word appears."
          },
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "errStream": 7.6,
            "ciStream": [
              5.4,
              10.1
            ],
            "finalsPerClip": 1.8,
            "tone": "warn",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50-chirp.json",
            "caveat": "Measured with the region-qualified `fr-FR` on the wire. The gateway sends the platform's bare `fr` until packages/providers deploys, and Google Speech v2 rejects it outright - so this row is reachable today only through the fixed code path. It was measured in the SAME job as the other thirteen vendors: the per-vendor wire code exists so this row is not ranked against measurements taken under different conditions. No batch cell: this column's batch series is the 2026-07-21 run and was not re-measured."
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 8.6,
            "ci": [
              5.6,
              12.3
            ],
            "errStream": 7.1,
            "ciStream": [
              4.9,
              9.4
            ],
            "finalsPerClip": 1,
            "tone": "warn",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "One of two rows that IMPROVE streamed (8.6% batch -> 7.1% streaming), which is what a realtime-first vendor looks like when batch is the handicap rather than the advantage. Commits once per clip and finalizes in 528ms."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 9.9,
            "ci": [
              7.9,
              12.1
            ],
            "errStream": 10.7,
            "ciStream": [
              8.2,
              13.4
            ],
            "finalsPerClip": 2.9,
            "tone": "warn",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "BIMODAL, not a wide tail, and REPRODUCED across two independent runs: 44 of 50 turns finalize around 200ms and 6 land in a tight 5210-5318ms band. That cluster is DEEPGRAM, not the audio - nova-2 and nova-3 are the only vendors those six clips affect, while all twelve others sit between 0.8x and 1.2x (AssemblyAI 3.1x is the sole partial exception). So 12% of French turns cost 5.2 seconds to end, and the 209ms median does not say so. p90 falls inside the cluster, which is why it reads 5210ms."
          },
          {
            "provider": "Gradium",
            "model": "Gradium ASR",
            "id": "gradium:default",
            "err": 10,
            "ci": [
              7.3,
              12.7
            ],
            "errStream": 11.8,
            "ciStream": [
              9.6,
              14.7
            ],
            "finalsPerClip": 16.4,
            "tone": "warn",
            "runDate": "2026-07-31",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "Re-measured 2026-07-31 on Gradium's new default model; the rest of the column is 2026-07-21. One saved raw run; no repeat-run determinism claim. Streaming measured 2026-08-21. It fragments the utterance into 16.4 finals per clip against 1.0-2.9 for every other row, and 0 of 50 clips came back perfect - the 1.8-point streaming loss is words dropped stitching those segments back together, not recognition.",
            "provenance": "apps/benchmark-runner/results/gradium-ml-fr-NEW.json",
            "flag": "Re-measured 2026-07-31 on Gradium's new default model; the rest of the column is 2026-07-21. One saved raw run; no repeat-run determinism claim."
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 10.4,
            "ci": [
              7.2,
              14.2
            ],
            "errStream": 12.5,
            "ciStream": [
              9.2,
              16.8
            ],
            "finalsPerClip": 2.2,
            "tone": "bad",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "One catastrophic clip; pooled WER includes it (batch median 5.6%). A fast and tight finalize - 293ms p50, 404ms p90, on 0 of 50 forced flushes, so it is a real vendor turn decision - paired with the slowest-but-one first token at 4923ms. The two Deepgram models beat it at the median, but only by trading a 5.2s second mode for it. Quick to decide the turn ended, slow to say anything before that. Worth stating because the same model force-flushed on 50 of 50 Spanish clips, where its published figure is our flush timeout rather than a vendor decision; French is not that case and the flush counts are what separate them."
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse",
            "id": "smallest:pulse",
            "errStream": 11.6,
            "ciStream": [
              8.2,
              15.4
            ],
            "finalsPerClip": 2.5,
            "tone": "bad",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "Reads 8.9% on the 42 clips whose reference carries no digits, against 11.6% over all 50 - a 2.7-point penalty that is scoring, not recognition. It is the only row on the column that writes French numerals as WORDS: \"mille neuf cent soixante-seize\" where the FLEURS reference says \"1976\", which the English-only normalizer cannot reconcile and charges as five insertions plus a substitution. It is also the only row that ever needed a forced flush: on 4 of 50 clips it made no turn decision and our timeout ended the turn at 8.2-8.3s, so its finalize latency is published over the other 46 (p90 984ms), and `--paired` drops those four from every vendor so none of them is ranked on a set that still contains them. Including them would have read 8221ms - our timeout under a vendor label."
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse Pro",
            "id": "smallest:pulse-pro",
            "err": 11.8,
            "ci": [
              8.5,
              15.8
            ],
            "tone": "bad",
            "caveat": "BATCH only - Pulse Pro is pre-recorded/HTTP and has no streaming endpoint, so the streaming row above is the sibling `pulse` model, not this one. The same digit-notation penalty applies to this cell and is quantified on that row; this number is therefore a floor, not a reading of what the model heard."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-2",
            "id": "deepgram:nova-2",
            "err": 12.9,
            "ci": [
              10.1,
              15.6
            ],
            "errStream": 13.3,
            "ciStream": [
              10.1,
              17.6
            ],
            "finalsPerClip": 2.9,
            "tone": "bad",
            "provenanceStream": "apps/benchmark-runner/results/fr-stream-wer-50.json",
            "caveat": "Two failure modes at once, and p90 shows neither. 6 of 50 finals land BEFORE the measured end of speech - the model is ending the turn early rather than the corpus misleading it, since the other vendors are positive on those clips. A separate 5 of 50 land in the same ~5.2s band nova-3 hits, on the same clips, which is what makes that cluster Deepgram behaviour rather than hard audio. p90 reads 339ms because the early finals pull the low end while the 5.2s cluster sits above the 90th percentile: read p95 for the turn that actually hurts."
          }
        ]
      },
      "ar": {
        "label": "Arabic",
        "metric": "CER",
        "meta": "FLEURS ar · mixed gateway/direct route · n=50 batch, n=47 streaming (paired across rows) · corpus CER (diacritic-normalized) · batch 2026-07-21 · streaming 2026-08-21",
        "path": "batch",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 2.7,
            "ci": [
              2.3,
              3.2
            ],
            "errStream": 3.3,
            "ciStream": [
              2.4,
              4.6
            ],
            "finalsPerClip": 1.3,
            "tone": "good"
          },
          {
            "provider": "AssemblyAI",
            "model": "Universal-3.5 Pro",
            "id": "assemblyai:universal-3-5-pro",
            "err": 3.3,
            "ci": [
              2.5,
              4.2
            ],
            "errStream": 3.5,
            "ciStream": [
              2.4,
              4.9
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "runDate": "2026-08-17",
            "caveat": "Measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro), not via the Speko gateway - the gateway /v1/transcribe one-shot returns empty finals for this vendor. Same model as the routable streaming pin, different transport. Run 2026-08-17; the rest of the column is older.",
            "provenance": "apps/benchmark-runner/results/ar-assemblyai-sync-fleurs50.json",
            "flag": "Two transports, one per cell: the BATCH figure is measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro) because the gateway /v1/transcribe one-shot returns empty finals for this vendor, while the STREAMING figure is the gateway WebSocket and needs no workaround. Same model on both. Batch run 2026-08-17."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "flag": "Batch only, by construction. STREAMING_STT_MODEL substitutes `qwen3-asr-flash-realtime` for any Alibaba pin, so this model never runs on the socket and has no streaming or end-of-turn number to publish — the realtime row below carries both.",
            "err": 3.5,
            "ci": [
              2.7,
              4.5
            ],
            "tone": "warn"
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR Realtime",
            "latencyKey": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 4.1,
            "ciStream": [
              3.1,
              5.4
            ],
            "finalsPerClip": 1.1,
            "tone": "warn",
            "runDate": "2026-08-21",
            "provenance": "apps/benchmark-runner/results/ar-stream-cer-50.json",
            "flag": "The realtime model an Alibaba-pinned streaming call is served, not the pinned `qwen3-asr-flash`. DashScope is Asia-hosted, so its end-of-turn carries a trans-Pacific RTT it cannot shed. No published pre-recorded endpoint, so no batch cell."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 4.2,
            "ciStream": [
              3.3,
              5.2
            ],
            "finalsPerClip": 1,
            "tone": "warn",
            "runDate": "2026-08-21",
            "provenance": "apps/benchmark-runner/results/ar-stream-cer-50.json",
            "flag": "Streaming path only — `scribe_v2_realtime` has no published pre-recorded endpoint, so the absent batch cell is a real state and not an unmeasured one. STREAMING_STT_MODEL serves it for any ElevenLabs pin."
          },
          {
            "provider": "xAI",
            "model": "Grok STT",
            "id": "xai:stt",
            "errStream": 4.7,
            "ciStream": [
              2.5,
              8.1
            ],
            "finalsPerClip": 1.1,
            "tone": "warn",
            "runDate": "2026-08-21",
            "provenance": "apps/benchmark-runner/results/ar-stream-cer-50.json",
            "flag": "Streaming path only. The batch arm is withheld board-wide pending the clean re-bench; this cell is the gateway WebSocket, 1.1 finals/clip, with per-row route provenance. No end-of-turn number: the endpointing probe reaches xAI only vendor-direct, and a vendor-direct latency next to gateway-routed rows would compare two transports."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 4.7,
            "ci": [
              3.1,
              6.8
            ],
            "errStream": 4.9,
            "ciStream": [
              3.4,
              7
            ],
            "finalsPerClip": 1.3,
            "tone": "warn"
          },
          {
            "provider": "Hamsa",
            "model": "S3",
            "id": "hamsa:s3",
            "err": 5,
            "ci": [
              3.5,
              6.9
            ],
            "tone": "warn",
            "runDate": "2026-08-16",
            "provenance": "apps/benchmark-runner/results/hamsa-ml-ar.json",
            "flag": "Measured 2026-08-16 on the same 50-clip FLEURS ar set and gateway path; the rest of the column is 2026-07-21. No streaming or end-of-turn number, and neither is a gap: Hamsa is turn-based whole-utterance (TURN_BASED_LIVE_STT_PROVIDERS in packages/core, deliberately outside STREAMING_STT_PROVIDERS), so the gateway byte-stream socket refuses it and the turn IS the utterance.",
            "finalizeMs": 390,
            "finalizeBasis": {
              "surface": "speko-gateway",
              "vantage": "dev-local (~100ms of client RTT above the us-east4 convention)",
              "measured": "2026-08-16",
              "n": 20,
              "note": "Warm-socket per-turn finalization p50 (p95 534ms), Arabic, run speko-stt-hamsa-s3-ar-2026-08-16. NOT the streaming endpoint finalize the vendor-direct sweep publishes, and the difference is the quantity rather than the vantage: Hamsa is turn-based whole-utterance, so the worker VAD-segments the call and hands over a complete turn, and this interval EXCLUDES the endpoint-detection wait every streaming row's finalize includes. The vendor-direct rig refuses this vendor class for that reason, so no swept reading will replace this one."
            },
            "perMinUsd": 0.0425
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 5.1,
            "ci": [
              3.2,
              8.3
            ],
            "errStream": 3.6,
            "ciStream": [
              2.9,
              4.4
            ],
            "finalsPerClip": 1,
            "tone": "warn"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 7.1,
            "ci": [
              6,
              8.4
            ],
            "errStream": 7.6,
            "ciStream": [
              6.1,
              9.6
            ],
            "finalsPerClip": 3.1,
            "tone": "bad"
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "flag": "No end-of-turn number, and it is withheld rather than unmeasured: on 30 of 30 clips, in two separate runs, the final arrived only when the audio stream ended. Cartesia turn detection lives on a separate /stt/turns endpoint we do not wire, so the probe measured itself running out of audio, not a turn decision. Also no cost cell — the English board prices `ink-2`, a different model.",
            "err": 10.2,
            "ci": [
              8.4,
              12
            ],
            "errStream": 10.1,
            "ciStream": [
              8.9,
              11.5
            ],
            "finalsPerClip": 3.1,
            "tone": "bad"
          }
        ]
      },
      "fil": {
        "label": "Filipino",
        "metric": "WER",
        "meta": "FLEURS fil_ph · via gateway · n=50 · corpus WER · 2026-08-21",
        "path": "batch",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 6.1,
            "ci": [
              4.6,
              8
            ],
            "errStream": 12.3,
            "ciStream": [
              9.3,
              16.3
            ],
            "finalsPerClip": 3.9,
            "tone": "good"
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2",
            "id": "elevenlabs:scribe_v2",
            "err": 7.2,
            "ci": [
              5.7,
              8.7
            ],
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "GPT Transcribe",
            "id": "openai:gpt-transcribe",
            "err": 9.5,
            "ci": [
              6.6,
              12.8
            ],
            "errStream": 13.4,
            "ciStream": [
              10.2,
              17
            ],
            "finalsPerClip": 4.2,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 10.8,
            "ci": [
              8.6,
              13.5
            ],
            "errStream": 17.5,
            "ciStream": [
              13.6,
              22.4
            ],
            "finalsPerClip": 4.3,
            "tone": "warn"
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 11.7,
            "ciStream": [
              9,
              15.2
            ],
            "finalsPerClip": 1,
            "tone": "warn",
            "caveat": "Reproduces the inherited speko-fil-stt-native-2026-07 reading (11.7%) exactly, from an independent run. It commits ONCE per clip, so the 4.5-point loss against batch Scribe v2 is recognition, not segmentation."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Live Transcribe",
            "id": "openai:gpt-live-transcribe",
            "errStream": 12.5,
            "ciStream": [
              9.3,
              16.5
            ],
            "finalsPerClip": 1,
            "tone": "warn"
          },
          {
            "provider": "Modulate",
            "model": "Velma 2 multilingual",
            "id": "modulate:velma-2-stt-streaming",
            "err": 12.5,
            "ci": [
              8.8,
              16.5
            ],
            "errStream": 6.2,
            "ciStream": [
              4.7,
              8
            ],
            "finalsPerClip": 1.2,
            "tone": "warn",
            "caveat": "The multilingual Velma 2 model, not the catalog default `velma-2-stt-streaming-english-v2`. Carries the vendor's 2026-08-07 'Invalid input audio' incident risk documented on the catalog row."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 18.8,
            "ci": [
              15,
              23.3
            ],
            "errStream": 29.7,
            "ciStream": [
              23.9,
              35.8
            ],
            "finalsPerClip": 2.1,
            "tone": "warn"
          },
          {
            "provider": "Alibaba",
            "model": "Qwen-ASR",
            "id": "alibaba:qwen-asr",
            "err": 18.8,
            "ci": [
              15,
              23.3
            ],
            "tone": "warn"
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 20.7,
            "ci": [
              16.3,
              26.2
            ],
            "errStream": 17.2,
            "ciStream": [
              14.3,
              20.7
            ],
            "finalsPerClip": 3.4,
            "tone": "bad"
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 21.9,
            "ci": [
              15.4,
              28.5
            ],
            "errStream": 8.9,
            "ciStream": [
              6.3,
              12.2
            ],
            "finalsPerClip": 1,
            "tone": "bad",
            "caveat": "The most level-sensitive row on the column after Chirp: 16.7% on the loud half against 26.5% on the quiet half."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 25.6,
            "ci": [
              21.9,
              29.5
            ],
            "errStream": 24.5,
            "ciStream": [
              21.1,
              28.5
            ],
            "finalsPerClip": 5.5,
            "tone": "bad"
          },
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 27.3,
            "ci": [
              20.2,
              35.4
            ],
            "errStream": 26.6,
            "ciStream": [
              19.4,
              34.3
            ],
            "finalsPerClip": 4.8,
            "tone": "bad",
            "caveat": "Loses 17.9 points between the loud and quiet halves of the corpus (17.9% against 35.8%) - the widest level sensitivity measured here, and a reason to avoid it for telephony even though it ranks mid-table on clean audio."
          }
        ]
      },
      "nb": {
        "label": "Norwegian",
        "metric": "WER",
        "meta": "FLEURS nb_no · via gateway · n=50 · corpus WER · batch 2026-08-21 · streaming 2026-07-21",
        "path": "batch",
        "rows": [
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2",
            "id": "elevenlabs:scribe_v2",
            "err": 5.6,
            "ci": [
              3.8,
              7.5
            ],
            "tone": "good"
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v1",
            "id": "elevenlabs:scribe_v1",
            "err": 5.7,
            "ci": [
              3.8,
              7.7
            ],
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 6.4,
            "ci": [
              4.6,
              8.3
            ],
            "tone": "good",
            "caveat": "Batch only. The gateway streaming adapter commits audio too early for this vendor and returns fragments, so no streaming cell is published for any OpenAI row on this column."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Transcribe",
            "id": "openai:gpt-transcribe",
            "err": 6.8,
            "ci": [
              5,
              8.6
            ],
            "tone": "good"
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 8.3,
            "ciStream": [
              6.3,
              10.5
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "flag": "Measured 2026-07-21 through the live gateway WebSocket; the ElevenLabs realtime adapter hardcoded scribe_v2_realtime."
          },
          {
            "provider": "Modulate",
            "model": "Velma 2 multilingual",
            "id": "modulate:velma-2-stt-streaming",
            "err": 8.4,
            "ci": [
              6.3,
              10.7
            ],
            "errStream": 13,
            "ciStream": [
              11,
              14.8
            ],
            "finalsPerClip": 1,
            "runDateStream": "2026-08-22",
            "provenanceStream": "apps/benchmark-runner/results/nb-stream-wer-modulate50.json",
            "finalizeMs": 577,
            "finalizeBasis": {
              "surface": "speko-gateway",
              "vantage": "GCP us-east4, 11 simultaneous gateway sockets",
              "measured": "2026-08-21",
              "n": 30,
              "p90Ms": 1237,
              "note": "Norwegian finalize p50 from the 2026-08-21 nb endpointing job, standing in for the whole model — the nb reading rather than the Filipino 402ms because fil is this board's one VAD-marker column. Gateway-internal, so not comparable to a vendor-direct sweep row. Read it against the per-language cells, which carry each language's OWN finalize (fil 402, ja 1428, nb 577): this model-level number is the nb one, and Japanese is 2.5x slower than it. Sweepable, unlike Hamsa — the english-v2 sibling is already in the sweep at 1207ms, so adding this pin retires the exception."
            },
            "perMinUsd": 0.001,
            "tone": "good",
            "caveat": "The multilingual Velma 2 model, not the catalog default `velma-2-stt-streaming-english-v2` — an English-named model must not carry a Norwegian cell. Carries the vendor's 2026-08-07 'Invalid input audio' incident risk documented on the catalog row."
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 8.6,
            "ci": [
              5.7,
              12.6
            ],
            "errStream": 8,
            "ciStream": [
              6.2,
              10
            ],
            "finalsPerClip": 1,
            "tone": "good"
          },
          {
            "provider": "AssemblyAI",
            "model": "Universal-3.5 Pro",
            "id": "assemblyai:universal-3-5-pro",
            "err": 10.1,
            "ci": [
              8,
              12.3
            ],
            "errStream": 12.3,
            "ciStream": [
              9.1,
              16
            ],
            "finalsPerClip": 1,
            "tone": "warn",
            "runDate": "2026-08-21",
            "provenanceStream": "apps/benchmark-runner/results/nb-assemblyai-u35pro-stream50.json",
            "runDateStream": "2026-08-17",
            "caveat": "Two transports on one row. STREAMING is this column’s own gateway WebSocket, run 2026-08-17 on the same 50 clips as the rest of the column (results/nb-assemblyai-u35pro-stream50.json). BATCH is the vendor Sync endpoint, run 2026-08-21, because the gateway one-shot returns empty finals for this vendor (results/nb-batch-fleurs50.json).",
            "provenance": "apps/benchmark-runner/results/nb-batch-fleurs50.json",
            "flag": "Two transports on one row. STREAMING is this column’s own gateway WebSocket, run 2026-08-17 on the same 50 clips as the rest of the column (results/nb-assemblyai-u35pro-stream50.json). BATCH is the vendor Sync endpoint, run 2026-08-21, because the gateway one-shot returns empty finals for this vendor (results/nb-batch-fleurs50.json)."
          },
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 11.2,
            "ci": [
              8.8,
              13.8
            ],
            "tone": "warn",
            "caveat": "Requested as `nb-NO` rather than the platform’s bare `nb`. Same Speko gateway, same `us` endpoint, same chirp_3 model as any other Chirp cell — Google Speech V2 rejects region-less codes and answers `The language \"no\" is not supported by the model \"chirp_3\" in the location named \"us\"`, which this column previously published as a coverage gap. Reachable as plain `nb` since the GOOGLE_STT_REGION_DEFAULTS alias deployed on 2026-08-22, verified against the live gateway; the reading is unchanged either way, being the same model on the same resolved code. Run 2026-08-21, 50/50 clips.",
            "provenance": "apps/benchmark-runner/results/nb-batch-chirp3-fleurs50.json",
            "flag": "Requested as `nb-NO` rather than the platform’s bare `nb`. Same Speko gateway, same `us` endpoint, same chirp_3 model as any other Chirp cell — Google Speech V2 rejects region-less codes and answers `The language \"no\" is not supported by the model \"chirp_3\" in the location named \"us\"`, which this column previously published as a coverage gap. Reachable as plain `nb` since the GOOGLE_STT_REGION_DEFAULTS alias deployed on 2026-08-22, verified against the live gateway; the reading is unchanged either way, being the same model on the same resolved code. Run 2026-08-21, 50/50 clips."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 12,
            "ci": [
              9.6,
              14.6
            ],
            "tone": "warn"
          },
          {
            "provider": "Alibaba",
            "model": "Qwen-ASR",
            "id": "alibaba:qwen-asr",
            "err": 12,
            "ci": [
              9.6,
              14.6
            ],
            "tone": "warn"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 14.4,
            "ci": [
              10.5,
              18.3
            ],
            "tone": "bad"
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR Realtime",
            "latencyKey": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 14.5,
            "ciStream": [
              11.8,
              17.7
            ],
            "finalsPerClip": 1,
            "tone": "bad",
            "caveat": "The realtime model an `alibaba:qwen3-asr-flash` pin actually reaches on the socket. Its end-of-turn latency was measured 2026-08-22 by sending Alibaba's own Norwegian code `no` — the socket answers `<400> InternalError.Algo.InvalidParameter: Language code 'nb'` to the platform's code, which is why this row published accuracy with no latency beside it until now. The alias deployed on 2026-08-22, so plain `nb` reaches it."
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 14.5,
            "ci": [
              12.2,
              17
            ],
            "tone": "bad",
            "caveat": "Requested as `no` rather than the platform’s `nb`. Same Speko gateway and same Ink-Whisper model — Cartesia keys Norwegian as `no` and answers `502 Invalid language` on `nb`, `nb-NO` and `nor`, which this column previously published as Cartesia having no Norwegian. Reachable as plain `nb` since the cartesiaSttLanguage alias deployed on 2026-08-22, verified against the live gateway; the reading is unchanged either way, being the same model on the same resolved code. Run 2026-08-21, 50/50 clips.",
            "provenance": "apps/benchmark-runner/results/nb-batch-cartesia-ink-whisper-fleurs50.json",
            "flag": "Requested as `no` rather than the platform’s `nb`. Same Speko gateway and same Ink-Whisper model — Cartesia keys Norwegian as `no` and answers `502 Invalid language` on `nb`, `nb-NO` and `nor`, which this column previously published as Cartesia having no Norwegian. Reachable as plain `nb` since the cartesiaSttLanguage alias deployed on 2026-08-22, verified against the live gateway; the reading is unchanged either way, being the same model on the same resolved code. Run 2026-08-21, 50/50 clips."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 14.6,
            "ci": [
              12.2,
              17.2
            ],
            "errStream": 14.9,
            "ciStream": [
              12.3,
              17.5
            ],
            "clipsStream": 45,
            "finalsPerClip": 3.2,
            "tone": "bad",
            "flag": "The STREAMING cell covers 45 of 50 clips: 5 returned no final transcript (transient stream drops). The batch cell scored all 50. This is also the most fragmented row on the column at 3.2 finals per clip, and it still lands within 0.3 points of its own batch reading."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-2",
            "id": "deepgram:nova-2",
            "err": 19.1,
            "ci": [
              16.4,
              22.1
            ],
            "errStream": 18.6,
            "ciStream": [
              15.4,
              22.1
            ],
            "clipsStream": 48,
            "finalsPerClip": 3.1,
            "tone": "bad",
            "flag": "The STREAMING cell covers 48 of 50 clips: 2 returned no final transcript. The batch cell scored all 50."
          }
        ]
      },
      "hi": {
        "label": "Hindi",
        "metric": "WER",
        "meta": "FLEURS hi_in · mixed gateway/direct route · n=50 · corpus WER · 2026-08-04",
        "path": "batch",
        "rows": [
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 6.1,
            "ci": [
              4.4,
              8.1
            ],
            "tone": "good",
            "caveat": "Measured direct against Google Speech V2 in the `eu` multi-region, not via the Speko gateway - our Google envelope resolves to `us`, where chirp_3 serves no Indic language. The saved output is labeled Google Chirp 3 but does not record a provider-returned model version.",
            "provenance": "apps/benchmark-runner/results/hi-fleurs50-chirp-eu.json",
            "runDate": "2026-08-04"
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 9.8,
            "ci": [
              7,
              12.9
            ],
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 12.3,
            "ci": [
              8,
              17.5
            ],
            "tone": "warn"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 12.7,
            "ci": [
              10,
              15.9
            ],
            "tone": "warn",
            "flag": "The mini model beats its own flagship on Hindi (21.0%) — the only language on this board where it does."
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse Pro",
            "id": "smallest:pulse-pro",
            "err": 13.4,
            "ci": [
              10.5,
              16.5
            ],
            "tone": "warn"
          },
          {
            "provider": "AssemblyAI",
            "model": "Universal-3.5 Pro",
            "id": "assemblyai:universal-3-5-pro",
            "err": 17.2,
            "ci": [
              14.6,
              20.3
            ],
            "tone": "warn",
            "runDate": "2026-08-17",
            "caveat": "Measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro), not via the Speko gateway - the gateway /v1/transcribe one-shot returns empty finals for this vendor. Same model as the routable streaming pin, different transport. Run 2026-08-17; the rest of the column is older.",
            "provenance": "apps/benchmark-runner/results/hi-assemblyai-sync-fleurs50.json",
            "flag": "Measured direct against AssemblyAI's Sync endpoint (universal-3-5-pro), not via the Speko gateway - the gateway /v1/transcribe one-shot returns empty finals for this vendor. Same model as the routable streaming pin, different transport. Run 2026-08-17; the rest of the column is older."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 20.3,
            "ci": [
              17.1,
              23.8
            ],
            "tone": "bad"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 21,
            "ci": [
              13,
              30.9
            ],
            "tone": "bad"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-2",
            "id": "deepgram:nova-2",
            "err": 25.1,
            "ci": [
              21.2,
              28.9
            ],
            "tone": "bad"
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 41.9,
            "ci": [
              37.3,
              46.9
            ],
            "tone": "bad"
          }
        ]
      },
      "ta": {
        "label": "Tamil",
        "metric": "CER",
        "meta": "FLEURS ta_in | mixed route | target n=50 | 1 invalid clip excluded | corpus CER | 2026-08-04",
        "path": "batch",
        "rows": [
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 7.4,
            "ci": [
              4.3,
              13
            ],
            "tone": "warn",
            "clips": 48,
            "comparison": {
              "metric": "WER",
              "err": 24.3
            },
            "caveat": "Measured direct against Google Speech V2 in the `eu` multi-region, not via the Speko gateway - our Google envelope resolves to `us`, where chirp_3 serves no Indic language. The saved output is labeled Google Chirp 3 but does not record a provider-returned model version. 48 scored clips: one request failure and one invalid corpus clip excluded.",
            "provenance": "apps/benchmark-runner/results/ta-fleurs50-chirp-eu.json",
            "runDate": "2026-08-04"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 10.1,
            "ci": [
              8.1,
              12.7
            ],
            "tone": "bad",
            "clips": 49,
            "comparison": {
              "metric": "WER",
              "err": 33.2
            }
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse Pro",
            "id": "smallest:pulse-pro",
            "err": 12.1,
            "ci": [
              6.2,
              19.3
            ],
            "tone": "bad",
            "clips": 48,
            "comparison": {
              "metric": "WER",
              "err": 26.4
            },
            "flag": "48 scored clips: one gateway failure and one invalid corpus clip excluded."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 14.2,
            "ci": [
              9.9,
              19.8
            ],
            "tone": "bad",
            "clips": 49,
            "comparison": {
              "metric": "WER",
              "err": 36.9
            }
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 14.6,
            "ci": [
              8.7,
              20.2
            ],
            "tone": "bad",
            "clips": 49,
            "comparison": {
              "metric": "WER",
              "err": 31.3
            }
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 16.1,
            "ci": [
              12.1,
              21.7
            ],
            "tone": "bad",
            "clips": 49,
            "comparison": {
              "metric": "WER",
              "err": 40.3
            }
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 30.6,
            "ci": [
              24.7,
              37.4
            ],
            "tone": "bad",
            "clips": 49,
            "comparison": {
              "metric": "WER",
              "err": 78.8
            }
          }
        ]
      },
      "te": {
        "label": "Telugu",
        "metric": "CER",
        "meta": "FLEURS te_in · mixed gateway/direct route · n=50 · corpus CER · 2026-08-04",
        "path": "batch",
        "rows": [
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 3.2,
            "ci": [
              2,
              4.8
            ],
            "tone": "good",
            "caveat": "Measured direct against Google Speech V2 in the `eu` multi-region, not via the Speko gateway - our Google envelope resolves to `us`, where chirp_3 serves no Indic language. The saved output is labeled Google Chirp 3 but does not record a provider-returned model version.",
            "provenance": "apps/benchmark-runner/results/te-fleurs50-chirp-eu.json",
            "runDate": "2026-08-04"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 7.5,
            "ci": [
              5.6,
              9.7
            ],
            "tone": "warn"
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse Pro",
            "id": "smallest:pulse-pro",
            "err": 8.5,
            "ci": [
              5.2,
              13.6
            ],
            "tone": "warn"
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 9.8,
            "ci": [
              4.6,
              15.9
            ],
            "tone": "warn"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o-mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 10.2,
            "ci": [
              6.9,
              14.9
            ],
            "tone": "bad"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 13.4,
            "ci": [
              9.4,
              18.4
            ],
            "tone": "bad"
          }
        ]
      },
      "ko": {
        "label": "Korean",
        "metric": "CER",
        "meta": "FLEURS ko_kr · live streaming sockets · n=50, CER paired over 48 · Finalize + TTFT from us-east4 · 2026-08-24",
        "path": "stream",
        "rows": [
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "errStream": 4.2,
            "ciStream": [
              2.4,
              6.2
            ],
            "tone": "good",
            "flag": "Semantic end-of-turn: 24 of 50 finals landed before speech end — the 2ms Finalize median is that mode; the slow mode sits at the 1.1s p90."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 4.5,
            "ciStream": [
              2.6,
              6.7
            ],
            "tone": "good",
            "flag": "Vendor-default 1.5s VAD commit dominates the Finalize number."
          },
          {
            "provider": "xAI",
            "model": "Grok STT",
            "latencyKey": "xai:stt",
            "errStream": 4.7,
            "ciStream": [
              2.7,
              7
            ],
            "tone": "good",
            "flag": "Language is auto-detected; the socket takes no hint."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "latencyKey": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 4.7,
            "ciStream": [
              2.5,
              7.4
            ],
            "tone": "good",
            "flag": "Asia-hosted endpoint - carries a trans-Pacific round trip it cannot shed."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "errStream": 4.9,
            "ciStream": [
              2.8,
              7.2
            ],
            "tone": "good"
          },
          {
            "provider": "Gladia",
            "model": "Solaria-1",
            "id": "gladia:solaria-1",
            "errStream": 5.9,
            "ciStream": [
              3.8,
              8.2
            ],
            "tone": "warn",
            "flag": "Vendor-default 0.05s endpointing re-segments; heavy Finalize tail."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Transcribe",
            "id": "openai:gpt-transcribe",
            "errStream": 6.4,
            "ciStream": [
              3.9,
              9.6
            ],
            "tone": "warn"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "errStream": 11.6,
            "ciStream": [
              8.5,
              15.2
            ],
            "tone": "bad"
          }
        ]
      },
      "zh": {
        "label": "Chinese (Mandarin)",
        "metric": "CER",
        "meta": "FLEURS cmn_hans_cn (validation split) · live streaming sockets · n=50, CER paired over 47 · Finalize + TTFT from us-east4 · 2026-08-24",
        "path": "stream",
        "rows": [
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "errStream": 5.5,
            "ciStream": [
              3.2,
              8
            ],
            "tone": "good",
            "flag": "Bimodal turn decision: Finalize p50 860ms against p90 2.3s."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "latencyKey": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 6,
            "ciStream": [
              3.5,
              8.8
            ],
            "tone": "good",
            "flag": "Asia-hosted endpoint - carries a trans-Pacific round trip it cannot shed."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 6.2,
            "ciStream": [
              3.9,
              8.8
            ],
            "tone": "good",
            "flag": "Vendor-default 1.5s VAD commit dominates the Finalize number."
          },
          {
            "provider": "xAI",
            "model": "Grok STT",
            "latencyKey": "xai:stt",
            "errStream": 6.7,
            "ciStream": [
              4.1,
              9.5
            ],
            "tone": "good",
            "flag": "Language is auto-detected; the socket takes no hint."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Transcribe",
            "id": "openai:gpt-transcribe",
            "errStream": 8.3,
            "ciStream": [
              5.5,
              11.1
            ],
            "tone": "warn"
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "errStream": 9.1,
            "ciStream": [
              6.2,
              12.3
            ],
            "tone": "warn"
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "errStream": 9.3,
            "ciStream": [
              6.8,
              12.1
            ],
            "tone": "warn"
          },
          {
            "provider": "Gladia",
            "model": "Solaria-1",
            "id": "gladia:solaria-1",
            "errStream": 10.3,
            "ciStream": [
              7.5,
              13.3
            ],
            "tone": "warn",
            "flag": "Vendor-default 0.05s endpointing re-segments; heavy Finalize tail."
          }
        ]
      },
      "ja": {
        "label": "Japanese",
        "metric": "CER",
        "meta": "FLEURS ja_jp · via gateway · n=50 · pooled CER over hiragana yomi · batch + streaming · Finalize + TTFT from US East (Virginia) · 2026-09-02",
        "path": "batch",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 1.6,
            "ci": [
              0.9,
              2.5
            ],
            "errStream": 5.2,
            "ciStream": [
              2.7,
              8.7
            ],
            "finalsPerClip": 2.4,
            "clipsStream": 49,
            "tone": "good",
            "flag": "Wins the batch path and loses 3.2 points on the socket — it fragments the utterance into 2.4 finals per clip (max 8) and the seams lose words: 群島や湖では came back as ぶんとう。 湖では。"
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2",
            "id": "elevenlabs:scribe_v2",
            "err": 2,
            "ci": [
              1,
              2.9
            ],
            "tone": "good",
            "flag": "Batch arm only. Any ElevenLabs pin sent to the socket is served by scribe_v2_realtime, which is a different model and gets its own row."
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 2,
            "ci": [
              1.1,
              2.8
            ],
            "errStream": 1.8,
            "ciStream": [
              1.1,
              2.5
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "flag": "The only arm that does not degrade streamed — exactly 1.00 finals per clip, one clean transcript, and it takes the streaming top spot."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v1",
            "id": "elevenlabs:scribe_v1",
            "err": 2.1,
            "ci": [
              1.1,
              3
            ],
            "tone": "good",
            "flag": "No streaming arm — scribe_v1 publishes no realtime endpoint."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 2.1,
            "ci": [
              1.1,
              3.3
            ],
            "errStream": 3.7,
            "ciStream": [
              2.1,
              5.8
            ],
            "finalsPerClip": 1.4,
            "clipsStream": 49,
            "tone": "good",
            "flag": "Streaming provenance returned qwen3-asr-flash, not the -realtime substitution the core registry usually swaps in for Alibaba pins."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen-ASR",
            "id": "alibaba:qwen-asr",
            "err": 2.1,
            "ci": [
              1.1,
              3.3
            ],
            "tone": "good",
            "flag": "Scores identically to Qwen3-ASR on every clip of this corpus — a paired bootstrap puts both in a tie with the batch leader (Δ +0.06pp, CI −1.05 to +1.31)."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Transcribe",
            "id": "openai:gpt-transcribe",
            "err": 2.8,
            "ci": [
              1.7,
              4
            ],
            "errStream": 3.7,
            "ciStream": [
              2.4,
              5.3
            ],
            "finalsPerClip": 2.6,
            "clipsStream": 47,
            "tone": "warn",
            "flag": "Three of 50 clips returned no final on the socket and are excluded from the streaming cell, not scored as deletions."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 2.8,
            "ci": [
              1.7,
              4
            ],
            "errStream": 4.5,
            "ciStream": [
              3.1,
              6
            ],
            "finalsPerClip": 2.5,
            "clipsStream": 46,
            "tone": "warn",
            "flag": "Significantly behind the batch leader on a paired test (Δ +1.66pp, CI +0.67 to +2.67) despite overlapping marginal intervals."
          },
          {
            "provider": "AssemblyAI",
            "model": "Universal-3.5 Pro",
            "latencyKey": "assemblyai:universal-3-5-pro",
            "errStream": 3.4,
            "ciStream": [
              2.1,
              4.9
            ],
            "finalsPerClip": 1.3,
            "tone": "warn",
            "flag": "Streaming only, and the missing batch cell is OUR harness: the gateway batch path answers `expected stream, got file` for every AssemblyAI selector. The vendor publishes ja and the socket serves universal-3-5-pro cleanly on all 50 clips."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 4.4,
            "ciStream": [
              2.6,
              7.1
            ],
            "finalsPerClip": 1.1,
            "tone": "warn",
            "flag": "This is what the socket serves for any ElevenLabs pin (STREAMING_STT_MODEL substitution), so the 2.4-point gap against Scribe v2 batch is a different model, not a path tax. Commits once per clip."
          },
          {
            "provider": "Modulate",
            "model": "Velma 2 multilingual",
            "err": 4.8,
            "ci": [
              3.4,
              6.4
            ],
            "errStream": 4.8,
            "ciStream": [
              3.5,
              6.5
            ],
            "finalsPerClip": 1.1,
            "tone": "warn",
            "flag": "The only arm whose two paths agree to the decimal."
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse Pro",
            "id": "smallest:pulse-pro",
            "err": 5.5,
            "ci": [
              3.7,
              7.8
            ],
            "tone": "warn",
            "flag": "Batch only — pulse-pro cannot drive the live socket, which serves `pulse` instead."
          },
          {
            "provider": "Smallest AI",
            "model": "Pulse",
            "latencyKey": "smallest:pulse",
            "errStream": 6.1,
            "ciStream": [
              4.1,
              8.4
            ],
            "finalsPerClip": 1.6,
            "tone": "warn",
            "flag": "The streaming substitution for any Smallest pin, so it is measured as its own model rather than under the Pulse Pro label. Its Finalize p90 is a FLUSH path, not turn detection: 18 of 50 finals arrived only after the stream was closed."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Live Transcribe",
            "errStream": 6.7,
            "ciStream": [
              3.1,
              11
            ],
            "finalsPerClip": 0.3,
            "clipsStream": 14,
            "tone": "bad",
            "flag": "RELIABILITY, not accuracy: 36 of 50 clips never produced a final (0.3 finals per clip). The 6.7% is over the 14 that survived, so it is a survivorship floor. Realtime-API-only model — the shortfall may be harness session config, and it is not a published vendor verdict."
          },
          {
            "provider": "Gladia",
            "model": "Solaria-1",
            "id": "gladia:solaria-1",
            "err": 11.5,
            "ci": [
              7.6,
              15.9
            ],
            "errStream": 17.6,
            "ciStream": [
              13.1,
              22.5
            ],
            "finalsPerClip": 2.5,
            "tone": "bad",
            "flag": "Run serialized at concurrency 1 — the staging key is a free-trial 1-session cap that 429s on any parallel run. One collapse and three short returns in 50 clips; conditional CER over the clean 46 is 9.3%."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 20.7,
            "ci": [
              17.1,
              24.5
            ],
            "errStream": 22,
            "ciStream": [
              18,
              26
            ],
            "finalsPerClip": 6.3,
            "tone": "bad",
            "flag": "Leads Finalize at 314 ms and is the accuracy floor of this column — a real speed/accuracy tradeoff, the same one Deepgram shows in Korean and Mandarin. Not a language-code artifact: re-running its three worst clips with the vendor multilingual hint (`multi`) made output worse and leaked Latin transliteration (Toya miでは). Zero collapses and zero truncations, but 6.3 finals per clip means the socket shreds the utterance."
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 30.3,
            "ci": [
              21.9,
              39
            ],
            "errStream": 35,
            "ciStream": [
              25.1,
              46.6
            ],
            "finalsPerClip": 1.5,
            "clipsStream": 46,
            "tone": "bad",
            "flag": "Degrades on 19 of 50 batch clips — 4 collapses (8%, Wilson 3–19%) plus 15 returns under 60% of the reference length, i.e. Whisper dropping the back half. Conditional CER over the intact clips is 12.5%."
          },
          {
            "provider": "xAI",
            "model": "Grok STT",
            "id": "xai:stt",
            "err": 42.9,
            "ci": [
              31.5,
              54.8
            ],
            "errStream": 34.6,
            "ciStream": [
              25.3,
              44.8
            ],
            "finalsPerClip": 1.5,
            "clipsStream": 48,
            "tone": "bad",
            "flag": "A FAILURE RATE, not a reading: 23 of 50 batch clips collapse (46%, Wilson 33–60%) to a single Chinese token (不, 就是), an English fragment (There's, Yes) or a 3-character shard. The 24 intact clips read 6.3%, but that subset is selected by the model's own output, so it is a survivorship floor. The identical collapse on the Turkish column was batch-path-only."
          }
        ]
      },
      "th": {
        "label": "Thai",
        "metric": "CER",
        "meta": "FLEURS th_th · via gateway (batch + streaming) · n=50 · pooled CER-th · n=50 streaming · Finalize + TTFT vendor-direct from US East (Virginia) and Asia Southeast (Singapore) · batch 2026-09-09 · streaming 2026-09-09",
        "path": "batch",
        "rows": [
          {
            "provider": "Alibaba",
            "model": "Qwen-ASR",
            "id": "alibaba:qwen-asr",
            "err": 3.5,
            "ci": [
              2.3,
              4.8
            ],
            "tone": "good",
            "flag": "Batch only. Scores within a hair of Qwen3-ASR on every clip and shares the batch top group; no socket runs this id."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Transcribe",
            "id": "openai:gpt-transcribe",
            "err": 3.5,
            "ci": [
              2.4,
              5
            ],
            "errStream": 4.3,
            "ciStream": [
              2.9,
              6
            ],
            "finalsPerClip": 2.4,
            "tone": "good",
            "flag": "Batch top group. Directionally worse streamed (16 of 21 moving clips, paired) but not significant after Holm correction; 2.4 finals per clip. Two of 50 clips returned no final on the socket and are excluded from the streaming cell, not scored as deletions."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR",
            "id": "alibaba:qwen3-asr-flash",
            "err": 3.5,
            "ci": [
              2.4,
              4.9
            ],
            "tone": "good",
            "flag": "BATCH ONLY. The streaming socket does not run this model: STREAMING_STT_MODEL in packages/core substitutes qwen3-asr-flash-realtime for any Alibaba pin, and the DashScope WS URL hardcodes it. Its streaming reading is the separate Qwen3-ASR Realtime row. The gateway still REPORTS model_id=qwen3-asr-flash on that socket, so the provenance stamp does not show the swap."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2",
            "id": "elevenlabs:scribe_v2",
            "err": 3.9,
            "ci": [
              2.5,
              5.7
            ],
            "tone": "good",
            "flag": "Batch arm only. Any ElevenLabs pin sent to the socket is served by scribe_v2_realtime, a different model with its own row. Writes Latin proper nouns (UNESCO, Sundarbans) where FLEURS spells them in Thai — a convention cost, not a collapse. 50/50 clips carry model_id=scribe_v2 on the file path."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v1",
            "id": "elevenlabs:scribe_v1",
            "err": 4.3,
            "ci": [
              2.8,
              6.1
            ],
            "tone": "good",
            "flag": "No streaming arm — scribe_v1 publishes no realtime endpoint. 50/50 clips carry model_id=scribe_v1, path=file, 50 distinct request ids, which is the bar that lets this row publish."
          },
          {
            "provider": "Alibaba",
            "model": "Qwen3-ASR Realtime",
            "id": "alibaba:qwen3-asr-flash-realtime",
            "errStream": 4.4,
            "ciStream": [
              3.1,
              6.1
            ],
            "finalsPerClip": 1.2,
            "tone": "good",
            "flag": "The model the socket actually runs for any Alibaba pin (STREAMING_STT_MODEL), now a catalog pin of its own. Singapore-homed: the only arm that gets FASTER from Singapore (finalize 861 ms Virginia → 659 ms, second-fastest there); its Virginia figure carries a trans-Pacific hop (TLS RTT 592 ms). One of 50 clips returned no final."
          },
          {
            "provider": "OpenAI",
            "model": "GPT Live Transcribe",
            "id": "openai:gpt-live-transcribe",
            "errStream": 4.5,
            "ciStream": [
              3.1,
              6.2
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "flag": "Realtime-API-only model, streaming only. This cell is a 1.0× realtime-pacing run: at the rig default 2.5× pacing it produced no final on 36 of 50 clips (every clip over ~8 s), a harness artifact, not a vendor result — 50 of 50 at real time. Commits once per clip."
          },
          {
            "provider": "Soniox",
            "model": "stt-rt-v5",
            "id": "soniox:stt-rt-v5",
            "err": 4.7,
            "ci": [
              3.2,
              6.7
            ],
            "errStream": 4.7,
            "ciStream": [
              3.2,
              6.8
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "flag": "The two paths agree to the decimal (4.7 / 4.7) and it commits exactly once per clip. Slow endpointer on Thai: 2.1 s finalize from Virginia, the same order as ElevenLabs."
          },
          {
            "provider": "ElevenLabs",
            "model": "Scribe v2 Realtime",
            "id": "elevenlabs:scribe_v2_realtime",
            "errStream": 4.9,
            "ciStream": [
              3.3,
              7
            ],
            "finalsPerClip": 1,
            "tone": "good",
            "flag": "This is what the socket serves for any ElevenLabs pin (STREAMING_STT_MODEL substitution), so the gap against Scribe v2 batch is a different model, not a path tax. Commits once per clip; 1.6 s finalize from Virginia is its ~1.5 s VAD default."
          },
          {
            "provider": "Modulate",
            "model": "Velma 2 multilingual",
            "id": "modulate:velma-2-stt-streaming",
            "err": 5.1,
            "ci": [
              3.2,
              7.3
            ],
            "errStream": 3.9,
            "ciStream": [
              2.6,
              5.2
            ],
            "finalsPerClip": 1,
            "tone": "warn",
            "flag": "Leads the streaming top group by point estimate — but the group is a 7-way CI tie and its own paired batch-to-stream delta is −0.7 points on 6 of 9 moving clips, not a gain. Commits exactly once per clip. Slowest finalize on the column at 2.9 s (Virginia)."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o mini Transcribe",
            "id": "openai:gpt-4o-mini-transcribe",
            "err": 5.2,
            "ci": [
              3.9,
              6.8
            ],
            "errStream": 6.9,
            "ciStream": [
              5.4,
              8.6
            ],
            "finalsPerClip": 2.4,
            "tone": "warn",
            "flag": "Directionally worse streamed (+1.9 points, 24 of 37 moving clips) — not significant after correction. 2.4 finals per clip. Two of 50 clips returned no final on the socket and are excluded from the streaming cell."
          },
          {
            "provider": "OpenAI",
            "model": "GPT-4o Transcribe",
            "id": "openai:gpt-4o-transcribe",
            "err": 6.1,
            "ci": [
              3.2,
              10.2
            ],
            "errStream": 5.7,
            "ciStream": [
              3.9,
              7.8
            ],
            "finalsPerClip": 2.4,
            "tone": "warn",
            "flag": "A WRONG-LANGUAGE RATE, not only a reading: 2 of 50 batch clips came back entirely in LAO script (Lao is Thai's sister language). Without them it is 3.6% CER — top group. Streamed, one Lao-mixed clip on a different clip, so the failure follows the model, not the path. 23 of 32 moving clips worse streamed (directional; not significant after Holm). Two no-final clips excluded from the streaming cell."
          },
          {
            "provider": "Google",
            "model": "Chirp 3",
            "id": "google:chirp_3",
            "err": 6.4,
            "ci": [
              4.4,
              9
            ],
            "errStream": 8.2,
            "ciStream": [
              5,
              12.8
            ],
            "finalsPerClip": 3.3,
            "tone": "warn",
            "flag": "Serves Thai from Google's `us` location (th-TH). THE ONE GATEWAY-PATH LATENCY ARM on this column — its finalize/TTFT ride through the staging gateway in us-central1 and pay a double backhaul from Singapore, so its Singapore delta is not comparable to the vendor-direct rows. Two finals committed more than 4 s before speech end (clips 0035/0036) and the same clips read short on the streaming WER check: a dropped tail, not a fast endpointer."
          },
          {
            "provider": "Gladia",
            "model": "Solaria-1",
            "id": "gladia:solaria-1",
            "err": 7.9,
            "ci": [
              5.9,
              10.4
            ],
            "errStream": 9.2,
            "ciStream": [
              6.9,
              12.5
            ],
            "finalsPerClip": 2.5,
            "tone": "warn",
            "flag": "Run serialized at concurrency 1 — the staging key is a free-trial 1-session cap. Directionally worse streamed (23 of 32 moving clips, raw p 0.02; not significant after Holm), 2.5 finals per clip. Two earlier serial passes each lost the same 2 of 50 finals; the published cell is a third pass that returned 50/50. Latency measured in its own job, so a different time window than the other arms."
          },
          {
            "provider": "Deepgram",
            "model": "Nova-3",
            "id": "deepgram:nova-3",
            "err": 8.4,
            "ci": [
              7.1,
              10
            ],
            "errStream": 8,
            "ciStream": [
              6.5,
              9.8
            ],
            "finalsPerClip": 3.6,
            "tone": "warn",
            "flag": "Leads finalize on both vantages (82 ms Virginia, 257 ms Singapore) and sits at the accuracy floor of the serious rows — the same speed/accuracy tradeoff it shows in Korean, Mandarin and Japanese. 3.6 finals per clip; 9 of 50 finals commit a few tens of ms before speech end (semantic-early, as on Korean)."
          },
          {
            "provider": "AssemblyAI",
            "model": "Universal (batch tier)",
            "err": 11.2,
            "ci": [
              8.8,
              14.1
            ],
            "tone": "bad",
            "flag": "BATCH ONLY, and the cell is the gateway's batch TIER, not a pinnable model: on /v1/transcribe the adapter maps every AssemblyAI pin to one pre-recorded model and all three routable pins returned byte-identical Thai. On the socket `universal-3-5-pro` answered Thai audio in Vietnamese (3/3 probe clips) — the 18-language tier has no Thai — so there is no streaming cell and no latency cell. Not routable."
          },
          {
            "provider": "xAI",
            "model": "Grok STT",
            "id": "xai:stt",
            "err": 16.6,
            "ci": [
              10,
              23.7
            ],
            "errStream": 9.8,
            "ciStream": [
              5.6,
              15.3
            ],
            "finalsPerClip": 1.2,
            "tone": "bad",
            "flag": "A TRUNCATION RATE on batch, not a reading: 8 of 50 batch clips return a tail fragment (สิ่งปลูกสร้างต่าง ๆ for a 79-character sentence); conditional CER over the intact 42 is 7.2%. Streamed it truncates 1 of 50 and reads 9.8% — the truncation is batch-path-only, as on Turkish and Japanese. One finalize turn was ended by our 8 s ceiling (flush_forced) and is excluded."
          },
          {
            "provider": "Cartesia",
            "model": "Ink-Whisper",
            "id": "cartesia:ink-whisper",
            "err": 24.1,
            "ci": [
              21.7,
              26.9
            ],
            "errStream": 25.1,
            "ciStream": [
              22.9,
              27.4
            ],
            "finalsPerClip": 2.6,
            "tone": "bad",
            "flag": "Honestly bad, not broken: garbled Thai with Latin proper nouns on both paths, 0 empties. Fast endpointer from Virginia (285 ms) but NEVER ENDPOINTED FROM SINGAPORE — 50 of 50 finals arrived only after our close (reproduced 20/20 on a replicate; 0/50 from Virginia), so its Singapore finalize is withheld and its TTFT (5.3 s, normal) stays."
          }
        ]
      }
    },
    "tts": {
      "es": {
        "label": "Spanish",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-07-21",
        "rows": [
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1951,
            "ci": [
              1859,
              2142
            ],
            "p": 0.9188,
            "tone": "good"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1751,
            "ci": [
              1662,
              1858
            ],
            "p": 0.7796,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v2",
            "elo": 1696,
            "ci": [
              1621,
              1785
            ],
            "p": 0.7134,
            "tone": "good",
            "estimated": true,
            "flag": "voice Emma · our own scorer, not the blind panel · +86 over the retiring tts-rt-v1 on the same scorer · the scorer reads 0.38 MOS high on Soniox Spanish · interval over the eight corpus sentences, not over raters"
          },
          {
            "provider": "xAI Grok",
            "model": "grok-tts",
            "id": "xai:tts",
            "elo": 1634,
            "ci": [
              1567,
              1718
            ],
            "p": 0.6601,
            "tone": "good"
          },
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1598,
            "ci": [
              1532,
              1671
            ],
            "p": 0.619,
            "tone": "warn"
          },
          {
            "provider": "Hume",
            "model": "octave-2",
            "id": "hume:octave-2",
            "elo": 1568,
            "ci": [
              1485,
              1655
            ],
            "p": 0.5826,
            "tone": "warn"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v1",
            "id": "soniox:tts-rt-v1",
            "elo": 1525,
            "ci": [
              1451,
              1608
            ],
            "p": 0.5302,
            "tone": "warn"
          },
          {
            "provider": "Smallest",
            "model": "lightning_v3.1",
            "id": "smallest:lightning_v3.1",
            "elo": 1487,
            "ci": [
              1410,
              1566
            ],
            "p": 0.4833,
            "tone": "warn"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.8-hd",
            "id": "minimax:speech-2.8-hd",
            "elo": 1460,
            "ci": [
              1396,
              1534
            ],
            "p": 0.4492,
            "tone": "warn"
          },
          {
            "provider": "Qwen",
            "model": "qwen3-tts-flash",
            "id": "alibaba:qwen3-tts-flash",
            "elo": 1437,
            "ci": [
              1384,
              1495
            ],
            "p": 0.422,
            "tone": "warn"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1408,
            "ci": [
              1355,
              1453
            ],
            "p": 0.3867,
            "tone": "warn"
          },
          {
            "provider": "Rime",
            "model": "arcanav3",
            "id": "rime:arcanav3",
            "elo": 1172,
            "ci": [
              1010,
              1271
            ],
            "p": 0.1574,
            "tone": "bad"
          },
          {
            "provider": "Deepgram",
            "model": "aura-2",
            "id": "deepgram:aura-2",
            "elo": 1055,
            "ci": [
              835,
              1181
            ],
            "p": 0.0852,
            "tone": "bad"
          }
        ]
      },
      "de": {
        "label": "German",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-07-21",
        "rows": [
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1798,
            "ci": [
              1724,
              1879
            ],
            "p": 0.8125,
            "tone": "good"
          },
          {
            "provider": "xAI Grok",
            "model": "grok-tts",
            "id": "xai:tts",
            "elo": 1762,
            "ci": [
              1666,
              1881
            ],
            "p": 0.7795,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v1",
            "id": "soniox:tts-rt-v1",
            "elo": 1717,
            "ci": [
              1638,
              1804
            ],
            "p": 0.7348,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v2",
            "elo": 1700,
            "ci": [
              1612,
              1784
            ],
            "p": 0.7053,
            "tone": "good",
            "estimated": true,
            "flag": "voice Emma · our own scorer, not the blind panel · it sits BELOW the panel tts-rt-v1 row because the scorer reads 0.51 MOS low on Soniox German, not because v2 regressed · +93 over v1 on the same scorer · interval over the eight corpus sentences, not over raters"
          },
          {
            "provider": "Hume",
            "model": "octave-2",
            "id": "hume:octave-2",
            "elo": 1687,
            "ci": [
              1623,
              1762
            ],
            "p": 0.7032,
            "tone": "good"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1665,
            "ci": [
              1592,
              1741
            ],
            "p": 0.6795,
            "tone": "good"
          },
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1576,
            "ci": [
              1480,
              1689
            ],
            "p": 0.5772,
            "tone": "warn"
          },
          {
            "provider": "Qwen",
            "model": "qwen3-tts-flash",
            "id": "alibaba:qwen3-tts-flash",
            "elo": 1573,
            "ci": [
              1507,
              1654
            ],
            "p": 0.5736,
            "tone": "warn"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.8-hd",
            "id": "minimax:speech-2.8-hd",
            "elo": 1528,
            "ci": [
              1470,
              1611
            ],
            "p": 0.5208,
            "tone": "warn"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1497,
            "ci": [
              1408,
              1586
            ],
            "p": 0.4835,
            "tone": "warn"
          },
          {
            "provider": "Rime",
            "model": "arcanav3",
            "id": "rime:arcanav3",
            "elo": 1257,
            "ci": [
              1121,
              1379
            ],
            "p": 0.2343,
            "tone": "bad"
          },
          {
            "provider": "Smallest",
            "model": "lightning_v3.1",
            "id": "smallest:lightning_v3.1",
            "elo": 1254,
            "ci": [
              1146,
              1333
            ],
            "p": 0.2315,
            "tone": "bad"
          },
          {
            "provider": "Gradium",
            "model": "Gradium TTS",
            "id": "gradium:default",
            "elo": 1245,
            "ci": [
              1114,
              1338
            ],
            "p": 0.2244,
            "tone": "bad"
          },
          {
            "provider": "Deepgram",
            "model": "aura-2",
            "id": "deepgram:aura-2",
            "elo": 1062,
            "ci": [
              912,
              1153
            ],
            "p": 0.0973,
            "tone": "bad"
          }
        ]
      },
      "fr": {
        "label": "French",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-07-21",
        "rows": [
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1775,
            "ci": [
              1691,
              1923
            ],
            "p": 0.8068,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v2",
            "elo": 1695,
            "ci": [
              1596,
              1798
            ],
            "p": 0.715,
            "tone": "good",
            "estimated": true,
            "flag": "voice Emma · our own scorer, not the blind panel · +78 over the retiring tts-rt-v1 on the same scorer · the scorer tracks the panel to 0.08 MOS on Soniox French · interval over the eight corpus sentences, not over raters"
          },
          {
            "provider": "xAI Grok",
            "model": "grok-tts",
            "id": "xai:tts",
            "elo": 1692,
            "ci": [
              1595,
              1799
            ],
            "p": 0.7242,
            "tone": "good"
          },
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1684,
            "ci": [
              1606,
              1770
            ],
            "p": 0.7154,
            "tone": "good"
          },
          {
            "provider": "Hume",
            "model": "octave-2",
            "id": "hume:octave-2",
            "elo": 1671,
            "ci": [
              1612,
              1762
            ],
            "p": 0.7004,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v1",
            "id": "soniox:tts-rt-v1",
            "elo": 1628,
            "ci": [
              1535,
              1739
            ],
            "p": 0.651,
            "tone": "good"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1622,
            "ci": [
              1533,
              1731
            ],
            "p": 0.6432,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1535,
            "ci": [
              1460,
              1631
            ],
            "p": 0.5365,
            "tone": "warn"
          },
          {
            "provider": "Qwen",
            "model": "qwen3-tts-flash",
            "id": "alibaba:qwen3-tts-flash",
            "elo": 1529,
            "ci": [
              1465,
              1615
            ],
            "p": 0.5285,
            "tone": "warn"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.8-hd",
            "id": "minimax:speech-2.8-hd",
            "elo": 1419,
            "ci": [
              1328,
              1503
            ],
            "p": 0.393,
            "tone": "warn"
          },
          {
            "provider": "Smallest",
            "model": "lightning_v3.1",
            "id": "smallest:lightning_v3.1",
            "elo": 1416,
            "ci": [
              1318,
              1491
            ],
            "p": 0.3896,
            "tone": "warn"
          },
          {
            "provider": "Rime",
            "model": "arcanav3",
            "id": "rime:arcanav3",
            "elo": 1316,
            "ci": [
              1215,
              1410
            ],
            "p": 0.2786,
            "tone": "bad"
          },
          {
            "provider": "Gradium",
            "model": "Gradium TTS",
            "id": "gradium:default",
            "elo": 1233,
            "ci": [
              1057,
              1321
            ],
            "p": 0.1999,
            "tone": "bad"
          },
          {
            "provider": "Deepgram",
            "model": "aura-2",
            "id": "deepgram:aura-2",
            "elo": 1114,
            "ci": [
              959,
              1196
            ],
            "p": 0.1123,
            "tone": "bad"
          }
        ]
      },
      "ar": {
        "label": "Arabic",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-07-21",
        "rows": [
          {
            "provider": "xAI Grok",
            "model": "grok-tts",
            "id": "xai:tts",
            "elo": 1736,
            "ci": [
              1656,
              1824
            ],
            "p": 0.7874,
            "tone": "good"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1644,
            "ci": [
              1542,
              1767
            ],
            "p": 0.6831,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v2",
            "elo": 1612,
            "ci": [
              1538,
              1699
            ],
            "p": 0.6259,
            "tone": "good",
            "estimated": true,
            "flag": "voice Emma · our own scorer, no human votes for Soniox in this language · +78 over the retiring tts-rt-v1, beaten on 8 of 8 lines · interval over the eight corpus sentences, not over raters"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1564,
            "ci": [
              1511,
              1621
            ],
            "p": 0.5831,
            "tone": "warn"
          },
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1454,
            "ci": [
              1383,
              1528
            ],
            "p": 0.4403,
            "tone": "warn"
          }
        ]
      },
      "fil": {
        "label": "Filipino",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-07-21",
        "rows": [
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 2012,
            "ci": [
              1854,
              2289
            ],
            "p": 0.9276,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v2",
            "elo": 1790,
            "ci": [
              1693,
              1923
            ],
            "p": 0.7659,
            "tone": "good",
            "estimated": true,
            "flag": "voice Emma · our own scorer, no human votes for Soniox in this language · +60 over the retiring tts-rt-v1, beaten on 8 of 8 lines · interval over the eight corpus sentences, not over raters"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1693,
            "ci": [
              1597,
              1815
            ],
            "p": 0.6925,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1676,
            "ci": [
              1552,
              1793
            ],
            "p": 0.6756,
            "tone": "good"
          },
          {
            "provider": "xAI Grok",
            "model": "grok-tts",
            "id": "xai:tts",
            "elo": 1657,
            "ci": [
              1534,
              1873
            ],
            "p": 0.6562,
            "tone": "good"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.8-hd",
            "id": "minimax:speech-2.8-hd",
            "elo": 1598,
            "ci": [
              1497,
              1717
            ],
            "p": 0.5943,
            "tone": "warn"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1597,
            "ci": [
              1509,
              1728
            ],
            "p": 0.5937,
            "tone": "warn"
          },
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1251,
            "ci": [
              1101,
              1346
            ],
            "p": 0.2538,
            "tone": "bad"
          }
        ]
      },
      "nb": {
        "label": "Norwegian",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-07-21",
        "rows": [
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1836,
            "ci": [
              1756,
              1949
            ],
            "p": 0.8724,
            "tone": "good"
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 1705,
            "ci": [
              1638,
              1790
            ],
            "p": 0.7519,
            "tone": "good"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1628,
            "ci": [
              1533,
              1733
            ],
            "p": 0.6674,
            "tone": "good"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v2",
            "elo": 1621,
            "ci": [
              1510,
              1743
            ],
            "p": 0.6384,
            "tone": "good",
            "estimated": true,
            "flag": "voice Emma · our own scorer, no human votes for Soniox in this language · +86 over the retiring tts-rt-v1, beaten on 8 of 8 lines · interval over the eight corpus sentences, not over raters"
          },
          {
            "provider": "Azure",
            "model": "Azure AI Speech (native)",
            "elo": 1467,
            "ci": [
              1391,
              1537
            ],
            "p": 0.4665,
            "tone": "warn"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1420,
            "ci": [
              1343,
              1482
            ],
            "p": 0.4053,
            "tone": "warn"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.8-hd",
            "id": "minimax:speech-2.8-hd",
            "elo": 1386,
            "ci": [
              1264,
              1487
            ],
            "p": 0.362,
            "tone": "bad"
          }
        ]
      },
      "hi": {
        "label": "Hindi",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-08-05",
        "rows": [
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1736,
            "ci": [
              1637,
              1903
            ],
            "p": 0.8197,
            "tone": "good"
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 1563,
            "ci": [
              1467,
              1683
            ],
            "p": 0.5977,
            "tone": "warn"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.8-hd",
            "id": "minimax:speech-2.8-hd",
            "elo": 1522,
            "ci": [
              1432,
              1636
            ],
            "p": 0.5355,
            "tone": "warn"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1490,
            "ci": [
              1362,
              1600
            ],
            "p": 0.4873,
            "tone": "warn"
          },
          {
            "provider": "Maya",
            "model": "Maya 2 Native",
            "elo": 1478,
            "ci": [
              1295,
              1666
            ],
            "p": 0.47,
            "tone": "warn",
            "flag": "voice Ananya · interval bootstrapped over the eight corpus sentences, not over raters — clip-to-clip spread is huge here (sd 1.90 on the scorer's own 1–5 output) — the widest interval on the board"
          },
          {
            "provider": "Google Chirp",
            "model": "chirp-3-hd",
            "elo": 1468,
            "ci": [
              1378,
              1557
            ],
            "p": 0.4547,
            "tone": "warn"
          },
          {
            "provider": "Azure",
            "model": "Azure AI Speech (native)",
            "elo": 1412,
            "ci": [
              1272,
              1515
            ],
            "p": 0.3715,
            "tone": "bad"
          },
          {
            "provider": "Speechify",
            "model": "simba-multilingual",
            "elo": 1310,
            "ci": [
              1134,
              1394
            ],
            "p": 0.2336,
            "tone": "bad"
          }
        ]
      },
      "ta": {
        "label": "Tamil",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-08-05",
        "rows": [
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 1676,
            "ci": [
              1590,
              1772
            ],
            "p": 0.7313,
            "tone": "good"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1620,
            "ci": [
              1565,
              1701
            ],
            "p": 0.6554,
            "tone": "good"
          },
          {
            "provider": "Maya",
            "model": "Maya 2 Native",
            "elo": 1603,
            "ci": [
              1541,
              1668
            ],
            "p": 0.63,
            "tone": "good",
            "flag": "voice Ananya · interval bootstrapped over the eight corpus sentences, not over raters — tightest of its three columns (sd 0.60 on that same 1–5 output) and it overlaps both Gemini above and Cartesia below"
          },
          {
            "provider": "Google Chirp",
            "model": "chirp-3-hd",
            "elo": 1555,
            "ci": [
              1507,
              1614
            ],
            "p": 0.56,
            "tone": "warn"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1543,
            "ci": [
              1484,
              1616
            ],
            "p": 0.5427,
            "tone": "warn"
          },
          {
            "provider": "Azure",
            "model": "Azure AI Speech (native)",
            "elo": 1462,
            "ci": [
              1364,
              1555
            ],
            "p": 0.4262,
            "tone": "warn"
          },
          {
            "provider": "Speechify",
            "model": "simba-multilingual",
            "elo": 1144,
            "ci": [
              933,
              1239
            ],
            "p": 0.0843,
            "tone": "bad"
          }
        ]
      },
      "te": {
        "label": "Telugu",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-08-05",
        "rows": [
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1639,
            "ci": [
              1558,
              1743
            ],
            "p": 0.706,
            "tone": "good"
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 1625,
            "ci": [
              1516,
              1743
            ],
            "p": 0.6852,
            "tone": "good"
          },
          {
            "provider": "Google Chirp",
            "model": "chirp-3-hd",
            "elo": 1548,
            "ci": [
              1472,
              1645
            ],
            "p": 0.5678,
            "tone": "warn"
          },
          {
            "provider": "Maya",
            "model": "Maya 2 Native",
            "elo": 1540,
            "ci": [
              1453,
              1625
            ],
            "p": 0.555,
            "tone": "warn",
            "flag": "voice Ananya · interval bootstrapped over the eight corpus sentences, not over raters — sd 0.92 across sentences on that 1–5 output; clear of Chirp3-HD below, under Cartesia above"
          },
          {
            "provider": "Azure",
            "model": "Azure AI Speech (native)",
            "elo": 1419,
            "ci": [
              1308,
              1524
            ],
            "p": 0.3698,
            "tone": "bad"
          },
          {
            "provider": "Speechify",
            "model": "simba-multilingual",
            "elo": 1269,
            "ci": [
              1137,
              1356
            ],
            "p": 0.1711,
            "tone": "bad"
          }
        ]
      },
      "zh": {
        "label": "Chinese (Mandarin)",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-08-24",
        "rows": [
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1688,
            "ci": [
              1629,
              1754
            ],
            "p": 0.738,
            "tone": "good",
            "flag": "voice Xiaoyin"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1675,
            "ci": [
              1587,
              1767
            ],
            "p": 0.7231,
            "tone": "good",
            "flag": "voice Bobo"
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 1649,
            "ci": [
              1586,
              1724
            ],
            "p": 0.6918,
            "tone": "good",
            "flag": "voice Kore"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1564,
            "ci": [
              1475,
              1659
            ],
            "p": 0.5814,
            "tone": "warn",
            "flag": "voice Jing"
          },
          {
            "provider": "Azure",
            "model": "Azure AI Speech (native)",
            "elo": 1552,
            "ci": [
              1470,
              1688
            ],
            "p": 0.5654,
            "tone": "warn",
            "flag": "voice Xiaoxiao"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.6-hd",
            "id": "minimax:speech-2.6-hd",
            "elo": 1467,
            "ci": [
              1411,
              1553
            ],
            "p": 0.4511,
            "tone": "warn",
            "flag": "voice Sweet Lady"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1365,
            "ci": [
              1263,
              1454
            ],
            "p": 0.3196,
            "tone": "bad",
            "flag": "voice nova"
          }
        ]
      },
      "ko": {
        "label": "Korean",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-08-24",
        "rows": [
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 1917,
            "ci": [
              1849,
              2035
            ],
            "p": 0.8839,
            "tone": "good",
            "flag": "voice Kore"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1859,
            "ci": [
              1767,
              1981
            ],
            "p": 0.8438,
            "tone": "good",
            "flag": "voice Seoyun"
          },
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1722,
            "ci": [
              1646,
              1821
            ],
            "p": 0.728,
            "tone": "good",
            "flag": "voice Minji"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3_conversational",
            "id": "elevenlabs:eleven_v3_conversational",
            "elo": 1628,
            "ci": [
              1565,
              1706
            ],
            "p": 0.6333,
            "tone": "good",
            "flag": "voice Sora"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1498,
            "ci": [
              1433,
              1576
            ],
            "p": 0.4908,
            "tone": "warn",
            "flag": "voice nova"
          },
          {
            "provider": "Azure",
            "model": "Azure AI Speech (native)",
            "elo": 1438,
            "ci": [
              1362,
              1498
            ],
            "p": 0.424,
            "tone": "warn",
            "flag": "voice SunHi"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.6-hd",
            "id": "minimax:speech-2.6-hd",
            "elo": 1407,
            "ci": [
              1339,
              1486
            ],
            "p": 0.3911,
            "tone": "warn",
            "flag": "voice CalmLady"
          },
          {
            "provider": "Rime",
            "model": "coda",
            "id": "rime:coda",
            "elo": 1187,
            "ci": [
              1105,
              1261
            ],
            "p": 0.1866,
            "tone": "bad",
            "flag": "voice astra — the router default; Rime ships no curated Korean voice"
          },
          {
            "provider": "AWS Polly",
            "model": "generative",
            "id": "polly:generative",
            "elo": 953,
            "ci": [
              696,
              1073
            ],
            "p": 0.0527,
            "tone": "bad",
            "flag": "voice Ruth — the router default; Polly ships no curated Korean voice"
          }
        ]
      },
      "ja": {
        "label": "Japanese",
        "meta": "Blind A/B listening study · arena Elo · 95% CI · 2026-09-01",
        "rows": [
          {
            "provider": "Gemini",
            "model": "gemini-3.1-flash-tts-preview",
            "id": "google-tts:gemini-3.1-flash-tts-preview",
            "elo": 1756,
            "ci": [
              1684,
              1857
            ],
            "p": 0.8134,
            "tone": "good",
            "flag": "voice Aoede"
          },
          {
            "provider": "ElevenLabs",
            "model": "eleven_v3",
            "id": "elevenlabs:eleven_v3",
            "elo": 1646,
            "ci": [
              1589,
              1726
            ],
            "p": 0.6912,
            "tone": "good",
            "flag": "native ja voice Rie"
          },
          {
            "provider": "Cartesia",
            "model": "sonic-3.5",
            "id": "cartesia:sonic-3.5",
            "elo": 1600,
            "ci": [
              1534,
              1687
            ],
            "p": 0.6339,
            "tone": "good",
            "flag": "native ja voice Aiko"
          },
          {
            "provider": "Inworld",
            "model": "inworld-tts-2",
            "id": "inworld:inworld-tts-2",
            "elo": 1597,
            "ci": [
              1542,
              1663
            ],
            "p": 0.63,
            "tone": "good",
            "flag": "native ja voice Asuka"
          },
          {
            "provider": "Azure",
            "model": "DragonHDLatestNeural",
            "id": "azure:DragonHDLatestNeural",
            "elo": 1539,
            "ci": [
              1478,
              1613
            ],
            "p": 0.5526,
            "tone": "warn",
            "flag": "native ja voice Nanami HD"
          },
          {
            "provider": "Soniox",
            "model": "tts-rt-v2",
            "id": "soniox:tts-rt-v2",
            "elo": 1447,
            "ci": [
              1391,
              1496
            ],
            "p": 0.4274,
            "tone": "warn",
            "flag": "voice Emma"
          },
          {
            "provider": "MiniMax",
            "model": "speech-2.6-hd",
            "id": "minimax:speech-2.6-hd",
            "elo": 1441,
            "ci": [
              1361,
              1521
            ],
            "p": 0.4185,
            "tone": "warn",
            "flag": "native ja voice CalmLady"
          },
          {
            "provider": "Google Chirp 3 HD",
            "model": "chirp-3-hd",
            "id": "google-chirp:chirp-3-hd",
            "elo": 1407,
            "ci": [
              1317,
              1477
            ],
            "p": 0.3737,
            "tone": "bad",
            "flag": "native ja voice Callirrhoe"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4o-mini-tts",
            "id": "openai:gpt-4o-mini-tts",
            "elo": 1338,
            "ci": [
              1247,
              1413
            ],
            "p": 0.2883,
            "tone": "bad",
            "flag": "voice coral"
          },
          {
            "provider": "Hume",
            "model": "octave-2",
            "id": "hume:octave-2",
            "elo": 1228,
            "ci": [
              1107,
              1304
            ],
            "p": 0.171,
            "tone": "bad",
            "flag": "voice Kora"
          }
        ]
      }
    },
    "llmTask": {
      "es": {
        "label": "Spanish",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 91.23,
            "taskScore": 91,
            "stalled": 15,
            "toolSilent": 92,
            "deadAir": 57,
            "fabrication": null,
            "adherence": 12,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 158,
            "nToolTurns": 98,
            "nCls": 68,
            "kStall": 8,
            "kDeadAir": 90,
            "kToolSilent": 90,
            "kAdherence": 68,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 90.35,
            "taskScore": 90,
            "stalled": 17,
            "toolSilent": 97,
            "deadAir": 60,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 156,
            "nToolTurns": 96,
            "nCls": 63,
            "kStall": 9,
            "kDeadAir": 93,
            "kToolSilent": 93,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 88.89,
            "taskScore": 89,
            "stalled": 20,
            "toolSilent": 69,
            "deadAir": 41,
            "fabrication": null,
            "adherence": 13,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 148,
            "nToolTurns": 88,
            "nCls": 87,
            "kStall": 11,
            "kDeadAir": 61,
            "kToolSilent": 61,
            "kAdherence": 87,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 85.96,
            "taskScore": 86,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 199,
            "nToolTurns": 140,
            "nCls": 59,
            "kStall": 9,
            "kDeadAir": 140,
            "kToolSilent": 140,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 85.09,
            "taskScore": 85,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 64,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 167,
            "nToolTurns": 107,
            "nCls": 58,
            "kStall": 13,
            "kDeadAir": 107,
            "kToolSilent": 107,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 83.77,
            "taskScore": 84,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 72,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 178,
            "nToolTurns": 124,
            "nCls": 54,
            "kStall": 9,
            "kDeadAir": 124,
            "kToolSilent": 124,
            "kAdherence": 54,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 83.33,
            "taskScore": 83,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 16,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 185,
            "nToolTurns": 128,
            "nCls": 57,
            "kStall": 9,
            "kDeadAir": 128,
            "kToolSilent": 128,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 81.58,
            "taskScore": 82,
            "stalled": 35,
            "toolSilent": 99,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 137,
            "nToolTurns": 77,
            "nCls": 61,
            "kStall": 19,
            "kDeadAir": 76,
            "kToolSilent": 76,
            "kAdherence": 61,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 79.39,
            "taskScore": 79,
            "stalled": 33,
            "toolSilent": 1,
            "deadAir": 1,
            "fabrication": null,
            "adherence": 91,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 146,
            "nToolTurns": 86,
            "nCls": 140,
            "kStall": 18,
            "kDeadAir": 1,
            "kToolSilent": 1,
            "kAdherence": 140,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 79.24,
            "taskScore": 79,
            "stalled": 37,
            "toolSilent": 100,
            "deadAir": 65,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 170,
            "nToolTurns": 110,
            "nCls": 60,
            "kStall": 20,
            "kDeadAir": 110,
            "kToolSilent": 110,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 76.17,
            "taskScore": 76,
            "stalled": 30,
            "toolSilent": 96,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 199,
            "nToolTurns": 147,
            "nCls": 56,
            "kStall": 16,
            "kDeadAir": 141,
            "kToolSilent": 141,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 76.02,
            "taskScore": 76,
            "stalled": 31,
            "toolSilent": 57,
            "deadAir": 39,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 166,
            "nToolTurns": 107,
            "nCls": 96,
            "kStall": 17,
            "kDeadAir": 64,
            "kToolSilent": 61,
            "kAdherence": 96,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 75.73,
            "taskScore": 76,
            "stalled": 26,
            "toolSilent": 21,
            "deadAir": 15,
            "fabrication": null,
            "adherence": 11,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 186,
            "nToolTurns": 130,
            "nCls": 155,
            "kStall": 14,
            "kDeadAir": 27,
            "kToolSilent": 27,
            "kAdherence": 99,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 74.56,
            "taskScore": 75,
            "stalled": 39,
            "toolSilent": 1,
            "deadAir": 1,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 156,
            "nToolTurns": 98,
            "nCls": 151,
            "kStall": 21,
            "kDeadAir": 1,
            "kToolSilent": 1,
            "kAdherence": 151,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 73.98,
            "taskScore": 74,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 73,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 187,
            "nToolTurns": 137,
            "nCls": 50,
            "kStall": 13,
            "kDeadAir": 137,
            "kToolSilent": 137,
            "kAdherence": 50,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 69.01,
            "taskScore": 69,
            "stalled": 61,
            "toolSilent": 100,
            "deadAir": 57,
            "fabrication": null,
            "adherence": 46,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 139,
            "nToolTurns": 79,
            "nCls": 46,
            "kStall": 33,
            "kDeadAir": 79,
            "kToolSilent": 79,
            "kAdherence": 46,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 63.16,
            "taskScore": 63,
            "stalled": 48,
            "toolSilent": 92,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 86,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 201,
            "nToolTurns": 144,
            "nCls": 66,
            "kStall": 26,
            "kDeadAir": 132,
            "kToolSilent": 132,
            "kAdherence": 66,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 31.87,
            "taskScore": 32,
            "stalled": 85,
            "toolSilent": 100,
            "deadAir": 37,
            "fabrication": null,
            "adherence": 93,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 95,
            "nToolTurns": 35,
            "nCls": 60,
            "kStall": 46,
            "kDeadAir": 35,
            "kToolSilent": 35,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 15,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 31.58,
            "taskScore": 32,
            "stalled": 13,
            "toolSilent": 100,
            "deadAir": 94,
            "fabrication": null,
            "adherence": 19,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 256,
            "nToolTurns": 240,
            "nCls": 16,
            "kStall": 7,
            "kDeadAir": 240,
            "kToolSilent": 240,
            "kAdherence": 16,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 12.43,
            "taskScore": 12,
            "stalled": 89,
            "toolSilent": 79,
            "deadAir": 15,
            "fabrication": null,
            "adherence": 90,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 74,
            "nToolTurns": 14,
            "nCls": 63,
            "kStall": 48,
            "kDeadAir": 11,
            "kToolSilent": 11,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 6,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "ar": {
        "label": "Arabic",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 96.49,
            "taskScore": 96,
            "stalled": 4,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 189,
            "nToolTurns": 129,
            "nCls": 60,
            "kStall": 2,
            "kDeadAir": 129,
            "kToolSilent": 129,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 91.23,
            "taskScore": 91,
            "stalled": 15,
            "toolSilent": 97,
            "deadAir": 60,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 157,
            "nToolTurns": 97,
            "nCls": 63,
            "kStall": 8,
            "kDeadAir": 94,
            "kToolSilent": 94,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 90.06,
            "taskScore": 90,
            "stalled": 13,
            "toolSilent": 91,
            "deadAir": 59,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 167,
            "nToolTurns": 108,
            "nCls": 69,
            "kStall": 7,
            "kDeadAir": 98,
            "kToolSilent": 98,
            "kAdherence": 68,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 87.13,
            "taskScore": 87,
            "stalled": 30,
            "toolSilent": 78,
            "deadAir": 45,
            "fabrication": null,
            "adherence": 27,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 141,
            "nToolTurns": 81,
            "nCls": 78,
            "kStall": 16,
            "kDeadAir": 63,
            "kToolSilent": 63,
            "kAdherence": 78,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 86.11,
            "taskScore": 86,
            "stalled": 22,
            "toolSilent": 100,
            "deadAir": 63,
            "fabrication": null,
            "adherence": 97,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 161,
            "nToolTurns": 101,
            "nCls": 60,
            "kStall": 12,
            "kDeadAir": 101,
            "kToolSilent": 101,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 83.77,
            "taskScore": 84,
            "stalled": 26,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 32,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 154,
            "nToolTurns": 94,
            "nCls": 154,
            "kStall": 14,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 154,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 82.75,
            "taskScore": 83,
            "stalled": 30,
            "toolSilent": 100,
            "deadAir": 64,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 169,
            "nToolTurns": 109,
            "nCls": 60,
            "kStall": 16,
            "kDeadAir": 109,
            "kToolSilent": 109,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 81.63,
            "taskScore": 82,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 4,
            "nTask": 49,
            "nFab": 0,
            "nStall": 46,
            "nTurns": 147,
            "nToolTurns": 99,
            "nCls": 48,
            "kStall": 9,
            "kDeadAir": 99,
            "kToolSilent": 99,
            "kAdherence": 48,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 80.7,
            "taskScore": 81,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 167,
            "nToolTurns": 110,
            "nCls": 57,
            "kStall": 11,
            "kDeadAir": 110,
            "kToolSilent": 110,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 80.7,
            "taskScore": 81,
            "stalled": 28,
            "toolSilent": 87,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 157,
            "nToolTurns": 97,
            "nCls": 70,
            "kStall": 15,
            "kDeadAir": 87,
            "kToolSilent": 84,
            "kAdherence": 70,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 79.68,
            "taskScore": 80,
            "stalled": 30,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 83,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 173,
            "nToolTurns": 117,
            "nCls": 56,
            "kStall": 16,
            "kDeadAir": 117,
            "kToolSilent": 117,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 78.87,
            "taskScore": 79,
            "stalled": 38,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 100,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 150,
            "nToolTurns": 94,
            "nCls": 150,
            "kStall": 20,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 150,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 78.51,
            "taskScore": 79,
            "stalled": 19,
            "toolSilent": 22,
            "deadAir": 17,
            "fabrication": null,
            "adherence": 5,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 202,
            "nToolTurns": 148,
            "nCls": 167,
            "kStall": 10,
            "kDeadAir": 35,
            "kToolSilent": 33,
            "kAdherence": 111,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 76.46,
            "taskScore": 76,
            "stalled": 26,
            "toolSilent": 100,
            "deadAir": 73,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 187,
            "nToolTurns": 136,
            "nCls": 51,
            "kStall": 14,
            "kDeadAir": 136,
            "kToolSilent": 136,
            "kAdherence": 51,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 71.93,
            "taskScore": 72,
            "stalled": 50,
            "toolSilent": 100,
            "deadAir": 53,
            "fabrication": null,
            "adherence": 93,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 128,
            "nToolTurns": 68,
            "nCls": 60,
            "kStall": 27,
            "kDeadAir": 68,
            "kToolSilent": 68,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 70.91,
            "taskScore": 71,
            "stalled": 39,
            "toolSilent": 92,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 81,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 195,
            "nToolTurns": 140,
            "nCls": 66,
            "kStall": 21,
            "kDeadAir": 129,
            "kToolSilent": 129,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 59.5,
            "taskScore": 60,
            "stalled": 72,
            "toolSilent": 100,
            "deadAir": 52,
            "fabrication": null,
            "adherence": 25,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 125,
            "nToolTurns": 65,
            "nCls": 60,
            "kStall": 39,
            "kDeadAir": 65,
            "kToolSilent": 65,
            "kAdherence": 35,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 52.63,
            "taskScore": 53,
            "stalled": 48,
            "toolSilent": 91,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 56,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 193,
            "nToolTurns": 141,
            "nCls": 64,
            "kStall": 26,
            "kDeadAir": 129,
            "kToolSilent": 129,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 25.88,
            "taskScore": 26,
            "stalled": 89,
            "toolSilent": 100,
            "deadAir": 32,
            "fabrication": null,
            "adherence": 85,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 88,
            "nToolTurns": 28,
            "nCls": 60,
            "kStall": 48,
            "kDeadAir": 28,
            "kToolSilent": 28,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 16,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 21.05,
            "taskScore": 21,
            "stalled": 19,
            "toolSilent": 100,
            "deadAir": 95,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 265,
            "nToolTurns": 252,
            "nCls": 13,
            "kStall": 10,
            "kDeadAir": 252,
            "kToolSilent": 252,
            "kAdherence": 13,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 11.84,
            "taskScore": 12,
            "stalled": 89,
            "toolSilent": 100,
            "deadAir": 20,
            "fabrication": null,
            "adherence": 87,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 75,
            "nToolTurns": 15,
            "nCls": 60,
            "kStall": 48,
            "kDeadAir": 15,
            "kToolSilent": 15,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 5,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "fil": {
        "label": "Filipino",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 89.47,
            "taskScore": 89,
            "stalled": 13,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 190,
            "nToolTurns": 130,
            "nCls": 59,
            "kStall": 7,
            "kDeadAir": 130,
            "kToolSilent": 130,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 88.6,
            "taskScore": 89,
            "stalled": 20,
            "toolSilent": 97,
            "deadAir": 59,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 152,
            "nToolTurns": 92,
            "nCls": 63,
            "kStall": 11,
            "kDeadAir": 89,
            "kToolSilent": 89,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 88.3,
            "taskScore": 88,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 174,
            "nToolTurns": 114,
            "nCls": 58,
            "kStall": 8,
            "kDeadAir": 114,
            "kToolSilent": 114,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 87.57,
            "taskScore": 88,
            "stalled": 15,
            "toolSilent": 90,
            "deadAir": 57,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 161,
            "nToolTurns": 101,
            "nCls": 70,
            "kStall": 8,
            "kDeadAir": 91,
            "kToolSilent": 91,
            "kAdherence": 70,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 85.38,
            "taskScore": 85,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 64,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 162,
            "nToolTurns": 103,
            "nCls": 59,
            "kStall": 11,
            "kDeadAir": 103,
            "kToolSilent": 103,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 83.77,
            "taskScore": 84,
            "stalled": 26,
            "toolSilent": 100,
            "deadAir": 61,
            "fabrication": null,
            "adherence": 93,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 154,
            "nToolTurns": 94,
            "nCls": 60,
            "kStall": 14,
            "kDeadAir": 94,
            "kToolSilent": 94,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 81.58,
            "taskScore": 82,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 63,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 160,
            "nToolTurns": 101,
            "nCls": 59,
            "kStall": 13,
            "kDeadAir": 101,
            "kToolSilent": 101,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 79.39,
            "taskScore": 79,
            "stalled": 35,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 9,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 144,
            "nToolTurns": 84,
            "nCls": 144,
            "kStall": 19,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 141,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 78.65,
            "taskScore": 79,
            "stalled": 28,
            "toolSilent": 56,
            "deadAir": 38,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 167,
            "nToolTurns": 108,
            "nCls": 104,
            "kStall": 15,
            "kDeadAir": 63,
            "kToolSilent": 61,
            "kAdherence": 104,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 77.68,
            "taskScore": 78,
            "stalled": 34,
            "toolSilent": 28,
            "deadAir": 19,
            "fabrication": null,
            "adherence": 10,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 180,
            "nToolTurns": 123,
            "nCls": 144,
            "kStall": 18,
            "kDeadAir": 34,
            "kToolSilent": 34,
            "kAdherence": 119,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 73.83,
            "taskScore": 74,
            "stalled": 46,
            "toolSilent": 84,
            "deadAir": 46,
            "fabrication": null,
            "adherence": 3,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 133,
            "nToolTurns": 73,
            "nCls": 72,
            "kStall": 25,
            "kDeadAir": 61,
            "kToolSilent": 61,
            "kAdherence": 72,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 17,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 73.83,
            "taskScore": 74,
            "stalled": 22,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 96,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 179,
            "nToolTurns": 126,
            "nCls": 53,
            "kStall": 12,
            "kDeadAir": 126,
            "kToolSilent": 126,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 73.25,
            "taskScore": 73,
            "stalled": 35,
            "toolSilent": 100,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 172,
            "nToolTurns": 116,
            "nCls": 56,
            "kStall": 19,
            "kDeadAir": 116,
            "kToolSilent": 116,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 69.74,
            "taskScore": 70,
            "stalled": 57,
            "toolSilent": 83,
            "deadAir": 42,
            "fabrication": null,
            "adherence": 46,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 123,
            "nToolTurns": 63,
            "nCls": 69,
            "kStall": 31,
            "kDeadAir": 52,
            "kToolSilent": 52,
            "kAdherence": 69,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 67.11,
            "taskScore": 67,
            "stalled": 46,
            "toolSilent": 94,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 96,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 181,
            "nToolTurns": 128,
            "nCls": 60,
            "kStall": 25,
            "kDeadAir": 120,
            "kToolSilent": 120,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 65.77,
            "taskScore": 66,
            "stalled": 47,
            "toolSilent": 96,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 60,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 191,
            "nToolTurns": 138,
            "nCls": 59,
            "kStall": 25,
            "kDeadAir": 132,
            "kToolSilent": 132,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 60.09,
            "taskScore": 60,
            "stalled": 74,
            "toolSilent": 100,
            "deadAir": 51,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 123,
            "nToolTurns": 63,
            "nCls": 37,
            "kStall": 40,
            "kDeadAir": 63,
            "kToolSilent": 63,
            "kAdherence": 37,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 54.68,
            "taskScore": 55,
            "stalled": 67,
            "toolSilent": 1,
            "deadAir": 1,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 137,
            "nToolTurns": 77,
            "nCls": 136,
            "kStall": 36,
            "kDeadAir": 1,
            "kToolSilent": 1,
            "kAdherence": 136,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 29.82,
            "taskScore": 30,
            "stalled": 50,
            "toolSilent": 100,
            "deadAir": 83,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 230,
            "nToolTurns": 190,
            "nCls": 40,
            "kStall": 27,
            "kDeadAir": 190,
            "kToolSilent": 190,
            "kAdherence": 40,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 27.92,
            "taskScore": 28,
            "stalled": 94,
            "toolSilent": 100,
            "deadAir": 42,
            "fabrication": null,
            "adherence": 62,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 104,
            "nToolTurns": 44,
            "nCls": 60,
            "kStall": 51,
            "kDeadAir": 44,
            "kToolSilent": 44,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 15,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 7.89,
            "taskScore": 8,
            "stalled": 94,
            "toolSilent": 100,
            "deadAir": 9,
            "fabrication": null,
            "adherence": 52,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 66,
            "nToolTurns": 6,
            "nCls": 60,
            "kStall": 51,
            "kDeadAir": 6,
            "kToolSilent": 6,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 2,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "de": {
        "label": "German",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 92.54,
            "taskScore": 93,
            "stalled": 17,
            "toolSilent": 75,
            "deadAir": 44,
            "fabrication": null,
            "adherence": 6,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 144,
            "nToolTurns": 84,
            "nCls": 81,
            "kStall": 9,
            "kDeadAir": 63,
            "kToolSilent": 63,
            "kAdherence": 81,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 91.23,
            "taskScore": 91,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 62,
            "fabrication": null,
            "adherence": 92,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 158,
            "nToolTurns": 98,
            "nCls": 60,
            "kStall": 8,
            "kDeadAir": 98,
            "kToolSilent": 98,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 89.47,
            "taskScore": 89,
            "stalled": 7,
            "toolSilent": 100,
            "deadAir": 72,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 202,
            "nToolTurns": 145,
            "nCls": 57,
            "kStall": 4,
            "kDeadAir": 145,
            "kToolSilent": 145,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 89.47,
            "taskScore": 89,
            "stalled": 11,
            "toolSilent": 84,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 173,
            "nToolTurns": 114,
            "nCls": 74,
            "kStall": 6,
            "kDeadAir": 96,
            "kToolSilent": 96,
            "kAdherence": 74,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 84.36,
            "taskScore": 84,
            "stalled": 28,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 172,
            "nToolTurns": 114,
            "nCls": 53,
            "kStall": 15,
            "kDeadAir": 114,
            "kToolSilent": 114,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 82.46,
            "taskScore": 82,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 182,
            "nToolTurns": 126,
            "nCls": 56,
            "kStall": 11,
            "kDeadAir": 126,
            "kToolSilent": 126,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 82.46,
            "taskScore": 82,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 72,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 181,
            "nToolTurns": 128,
            "nCls": 53,
            "kStall": 8,
            "kDeadAir": 128,
            "kToolSilent": 128,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 82.16,
            "taskScore": 82,
            "stalled": 22,
            "toolSilent": 70,
            "deadAir": 47,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 172,
            "nToolTurns": 113,
            "nCls": 89,
            "kStall": 12,
            "kDeadAir": 81,
            "kToolSilent": 79,
            "kAdherence": 89,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 81.58,
            "taskScore": 82,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 178,
            "nToolTurns": 121,
            "nCls": 57,
            "kStall": 11,
            "kDeadAir": 121,
            "kToolSilent": 121,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 81.58,
            "taskScore": 82,
            "stalled": 31,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 173,
            "nToolTurns": 115,
            "nCls": 58,
            "kStall": 17,
            "kDeadAir": 115,
            "kToolSilent": 115,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 80.56,
            "taskScore": 81,
            "stalled": 31,
            "toolSilent": 1,
            "deadAir": 1,
            "fabrication": null,
            "adherence": 72,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 149,
            "nToolTurns": 89,
            "nCls": 148,
            "kStall": 17,
            "kDeadAir": 1,
            "kToolSilent": 1,
            "kAdherence": 148,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 79.97,
            "taskScore": 80,
            "stalled": 30,
            "toolSilent": 8,
            "deadAir": 6,
            "fabrication": null,
            "adherence": 9,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 179,
            "nToolTurns": 121,
            "nCls": 162,
            "kStall": 16,
            "kDeadAir": 11,
            "kToolSilent": 10,
            "kAdherence": 119,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 77.53,
            "taskScore": 78,
            "stalled": 36,
            "toolSilent": 3,
            "deadAir": 2,
            "fabrication": null,
            "adherence": 100,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 152,
            "nToolTurns": 93,
            "nCls": 148,
            "kStall": 19,
            "kDeadAir": 3,
            "kToolSilent": 3,
            "kAdherence": 148,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 76.32,
            "taskScore": 76,
            "stalled": 46,
            "toolSilent": 89,
            "deadAir": 48,
            "fabrication": null,
            "adherence": 81,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 130,
            "nToolTurns": 70,
            "nCls": 64,
            "kStall": 25,
            "kDeadAir": 62,
            "kToolSilent": 62,
            "kAdherence": 64,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 73.54,
            "taskScore": 74,
            "stalled": 39,
            "toolSilent": 94,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 204,
            "nToolTurns": 154,
            "nCls": 56,
            "kStall": 21,
            "kDeadAir": 144,
            "kToolSilent": 144,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 72.66,
            "taskScore": 73,
            "stalled": 26,
            "toolSilent": 100,
            "deadAir": 72,
            "fabrication": null,
            "adherence": 94,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 195,
            "nToolTurns": 141,
            "nCls": 54,
            "kStall": 14,
            "kDeadAir": 141,
            "kToolSilent": 141,
            "kAdherence": 54,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 69.59,
            "taskScore": 70,
            "stalled": 39,
            "toolSilent": 90,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 80,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 206,
            "nToolTurns": 155,
            "nCls": 65,
            "kStall": 21,
            "kDeadAir": 140,
            "kToolSilent": 140,
            "kAdherence": 65,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 64.91,
            "taskScore": 65,
            "stalled": 65,
            "toolSilent": 99,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 135,
            "nToolTurns": 75,
            "nCls": 42,
            "kStall": 35,
            "kDeadAir": 74,
            "kToolSilent": 74,
            "kAdherence": 42,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 48.25,
            "taskScore": 48,
            "stalled": 31,
            "toolSilent": 100,
            "deadAir": 79,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 212,
            "nToolTurns": 168,
            "nCls": 44,
            "kStall": 17,
            "kDeadAir": 168,
            "kToolSilent": 168,
            "kAdherence": 44,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 19.3,
            "taskScore": 19,
            "stalled": 89,
            "toolSilent": 100,
            "deadAir": 30,
            "fabrication": null,
            "adherence": 77,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 86,
            "nToolTurns": 26,
            "nCls": 60,
            "kStall": 48,
            "kDeadAir": 26,
            "kToolSilent": 26,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 11,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 15.94,
            "taskScore": 16,
            "stalled": 85,
            "toolSilent": 94,
            "deadAir": 20,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 76,
            "nToolTurns": 16,
            "nCls": 61,
            "kStall": 46,
            "kDeadAir": 15,
            "kToolSilent": 15,
            "kAdherence": 61,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 6,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "fr": {
        "label": "French",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 92.11,
            "taskScore": 92,
            "stalled": 13,
            "toolSilent": 97,
            "deadAir": 60,
            "fabrication": null,
            "adherence": 87,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 159,
            "nToolTurns": 99,
            "nCls": 63,
            "kStall": 7,
            "kDeadAir": 96,
            "kToolSilent": 96,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 89.47,
            "taskScore": 89,
            "stalled": 19,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 174,
            "nToolTurns": 115,
            "nCls": 174,
            "kStall": 10,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 174,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 89.47,
            "taskScore": 89,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 198,
            "nToolTurns": 138,
            "nCls": 60,
            "kStall": 8,
            "kDeadAir": 138,
            "kToolSilent": 138,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 88.74,
            "taskScore": 89,
            "stalled": 20,
            "toolSilent": 69,
            "deadAir": 41,
            "fabrication": null,
            "adherence": 5,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 150,
            "nToolTurns": 90,
            "nCls": 88,
            "kStall": 11,
            "kDeadAir": 62,
            "kToolSilent": 62,
            "kAdherence": 88,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 85.96,
            "taskScore": 86,
            "stalled": 19,
            "toolSilent": 89,
            "deadAir": 57,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 164,
            "nToolTurns": 105,
            "nCls": 71,
            "kStall": 10,
            "kDeadAir": 93,
            "kToolSilent": 93,
            "kAdherence": 71,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 85.67,
            "taskScore": 86,
            "stalled": 22,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 76,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 153,
            "nToolTurns": 93,
            "nCls": 153,
            "kStall": 12,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 153,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 84.8,
            "taskScore": 85,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 178,
            "nToolTurns": 121,
            "nCls": 57,
            "kStall": 9,
            "kDeadAir": 121,
            "kToolSilent": 121,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 81.58,
            "taskScore": 82,
            "stalled": 33,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 175,
            "nToolTurns": 116,
            "nCls": 59,
            "kStall": 18,
            "kDeadAir": 116,
            "kToolSilent": 116,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 81.43,
            "taskScore": 81,
            "stalled": 30,
            "toolSilent": 100,
            "deadAir": 63,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 161,
            "nToolTurns": 101,
            "nCls": 58,
            "kStall": 16,
            "kDeadAir": 101,
            "kToolSilent": 101,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 80.99,
            "taskScore": 81,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 87,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 176,
            "nToolTurns": 122,
            "nCls": 51,
            "kStall": 13,
            "kDeadAir": 122,
            "kToolSilent": 122,
            "kAdherence": 51,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 80.56,
            "taskScore": 81,
            "stalled": 20,
            "toolSilent": 12,
            "deadAir": 8,
            "fabrication": null,
            "adherence": 25,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 193,
            "nToolTurns": 135,
            "nCls": 174,
            "kStall": 11,
            "kDeadAir": 16,
            "kToolSilent": 16,
            "kAdherence": 153,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 79.82,
            "taskScore": 80,
            "stalled": 39,
            "toolSilent": 96,
            "deadAir": 54,
            "fabrication": null,
            "adherence": 84,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 137,
            "nToolTurns": 77,
            "nCls": 62,
            "kStall": 21,
            "kDeadAir": 74,
            "kToolSilent": 74,
            "kAdherence": 62,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 78.07,
            "taskScore": 78,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 175,
            "nToolTurns": 118,
            "nCls": 57,
            "kStall": 13,
            "kDeadAir": 118,
            "kToolSilent": 118,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 77.63,
            "taskScore": 78,
            "stalled": 26,
            "toolSilent": 63,
            "deadAir": 45,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 166,
            "nToolTurns": 106,
            "nCls": 92,
            "kStall": 14,
            "kDeadAir": 74,
            "kToolSilent": 67,
            "kAdherence": 92,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 73.25,
            "taskScore": 73,
            "stalled": 26,
            "toolSilent": 100,
            "deadAir": 73,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 185,
            "nToolTurns": 135,
            "nCls": 50,
            "kStall": 14,
            "kDeadAir": 135,
            "kToolSilent": 135,
            "kAdherence": 50,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 68.27,
            "taskScore": 68,
            "stalled": 61,
            "toolSilent": 95,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 11,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 143,
            "nToolTurns": 83,
            "nCls": 38,
            "kStall": 33,
            "kDeadAir": 79,
            "kToolSilent": 79,
            "kAdherence": 38,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 67.4,
            "taskScore": 67,
            "stalled": 50,
            "toolSilent": 98,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 185,
            "nToolTurns": 127,
            "nCls": 55,
            "kStall": 27,
            "kDeadAir": 125,
            "kToolSilent": 125,
            "kAdherence": 55,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 63.6,
            "taskScore": 64,
            "stalled": 39,
            "toolSilent": 86,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 80,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 209,
            "nToolTurns": 159,
            "nCls": 69,
            "kStall": 21,
            "kDeadAir": 137,
            "kToolSilent": 137,
            "kAdherence": 69,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 31.58,
            "taskScore": 32,
            "stalled": 22,
            "toolSilent": 100,
            "deadAir": 92,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 247,
            "nToolTurns": 227,
            "nCls": 20,
            "kStall": 12,
            "kDeadAir": 227,
            "kToolSilent": 227,
            "kAdherence": 20,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 26.32,
            "taskScore": 26,
            "stalled": 91,
            "toolSilent": 100,
            "deadAir": 38,
            "fabrication": null,
            "adherence": 83,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 96,
            "nToolTurns": 36,
            "nCls": 60,
            "kStall": 49,
            "kDeadAir": 36,
            "kToolSilent": 36,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 15,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 24.12,
            "taskScore": 24,
            "stalled": 83,
            "toolSilent": 78,
            "deadAir": 24,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 87,
            "nToolTurns": 27,
            "nCls": 66,
            "kStall": 45,
            "kDeadAir": 21,
            "kToolSilent": 21,
            "kAdherence": 66,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 9,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "nb": {
        "label": "Norwegian",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 92.98,
            "taskScore": 93,
            "stalled": 11,
            "toolSilent": 100,
            "deadAir": 63,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 162,
            "nToolTurns": 102,
            "nCls": 60,
            "kStall": 6,
            "kDeadAir": 102,
            "kToolSilent": 102,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 90.35,
            "taskScore": 90,
            "stalled": 13,
            "toolSilent": 100,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 184,
            "nToolTurns": 124,
            "nCls": 60,
            "kStall": 7,
            "kDeadAir": 124,
            "kToolSilent": 124,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 89.18,
            "taskScore": 89,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 64,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 169,
            "nToolTurns": 109,
            "nCls": 60,
            "kStall": 8,
            "kDeadAir": 109,
            "kToolSilent": 109,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 86.26,
            "taskScore": 86,
            "stalled": 15,
            "toolSilent": 85,
            "deadAir": 56,
            "fabrication": null,
            "adherence": 1,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 172,
            "nToolTurns": 113,
            "nCls": 70,
            "kStall": 8,
            "kDeadAir": 96,
            "kToolSilent": 96,
            "kAdherence": 70,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 84.06,
            "taskScore": 84,
            "stalled": 33,
            "toolSilent": 85,
            "deadAir": 48,
            "fabrication": null,
            "adherence": 5,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 138,
            "nToolTurns": 78,
            "nCls": 72,
            "kStall": 18,
            "kDeadAir": 66,
            "kToolSilent": 66,
            "kAdherence": 72,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 83.04,
            "taskScore": 83,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 180,
            "nToolTurns": 123,
            "nCls": 57,
            "kStall": 9,
            "kDeadAir": 123,
            "kToolSilent": 123,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 82.75,
            "taskScore": 83,
            "stalled": 20,
            "toolSilent": 81,
            "deadAir": 54,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 172,
            "nToolTurns": 113,
            "nCls": 78,
            "kStall": 11,
            "kDeadAir": 93,
            "kToolSilent": 91,
            "kAdherence": 78,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 81.29,
            "taskScore": 81,
            "stalled": 30,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 180,
            "nToolTurns": 123,
            "nCls": 57,
            "kStall": 16,
            "kDeadAir": 123,
            "kToolSilent": 123,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 81.1,
            "taskScore": 81,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 0,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 179,
            "nToolTurns": 124,
            "nCls": 55,
            "kStall": 8,
            "kDeadAir": 124,
            "kToolSilent": 124,
            "kAdherence": 55,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 79.39,
            "taskScore": 79,
            "stalled": 31,
            "toolSilent": 1,
            "deadAir": 1,
            "fabrication": null,
            "adherence": 67,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 147,
            "nToolTurns": 87,
            "nCls": 145,
            "kStall": 17,
            "kDeadAir": 1,
            "kToolSilent": 1,
            "kAdherence": 145,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 78.07,
            "taskScore": 78,
            "stalled": 39,
            "toolSilent": 99,
            "deadAir": 54,
            "fabrication": null,
            "adherence": 54,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 133,
            "nToolTurns": 73,
            "nCls": 61,
            "kStall": 21,
            "kDeadAir": 72,
            "kToolSilent": 72,
            "kAdherence": 61,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 75.73,
            "taskScore": 76,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 73,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 183,
            "nToolTurns": 130,
            "nCls": 51,
            "kStall": 13,
            "kDeadAir": 130,
            "kToolSilent": 130,
            "kAdherence": 51,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 73.98,
            "taskScore": 74,
            "stalled": 24,
            "toolSilent": 24,
            "deadAir": 17,
            "fabrication": null,
            "adherence": 8,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 193,
            "nToolTurns": 138,
            "nCls": 152,
            "kStall": 13,
            "kDeadAir": 33,
            "kToolSilent": 33,
            "kAdherence": 99,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 73.68,
            "taskScore": 74,
            "stalled": 13,
            "toolSilent": 100,
            "deadAir": 78,
            "fabrication": null,
            "adherence": 91,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 206,
            "nToolTurns": 160,
            "nCls": 46,
            "kStall": 7,
            "kDeadAir": 160,
            "kToolSilent": 160,
            "kAdherence": 46,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 67.98,
            "taskScore": 68,
            "stalled": 48,
            "toolSilent": 98,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 96,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 178,
            "nToolTurns": 124,
            "nCls": 54,
            "kStall": 26,
            "kDeadAir": 122,
            "kToolSilent": 122,
            "kAdherence": 54,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 66.96,
            "taskScore": 67,
            "stalled": 52,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 146,
            "nToolTurns": 86,
            "nCls": 142,
            "kStall": 28,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 142,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 65.79,
            "taskScore": 66,
            "stalled": 63,
            "toolSilent": 100,
            "deadAir": 56,
            "fabrication": null,
            "adherence": 22,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 137,
            "nToolTurns": 77,
            "nCls": 41,
            "kStall": 34,
            "kDeadAir": 77,
            "kToolSilent": 77,
            "kAdherence": 41,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 65.79,
            "taskScore": 66,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 74,
            "fabrication": null,
            "adherence": 4,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 201,
            "nToolTurns": 148,
            "nCls": 53,
            "kStall": 13,
            "kDeadAir": 148,
            "kToolSilent": 148,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 58.77,
            "taskScore": 59,
            "stalled": 48,
            "toolSilent": 90,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 41,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 199,
            "nToolTurns": 148,
            "nCls": 62,
            "kStall": 26,
            "kDeadAir": 133,
            "kToolSilent": 133,
            "kAdherence": 62,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 26.75,
            "taskScore": 27,
            "stalled": 89,
            "toolSilent": 100,
            "deadAir": 36,
            "fabrication": null,
            "adherence": 74,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 94,
            "nToolTurns": 34,
            "nCls": 60,
            "kStall": 48,
            "kDeadAir": 34,
            "kToolSilent": 34,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 15,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 6.58,
            "taskScore": 7,
            "stalled": 98,
            "toolSilent": 89,
            "deadAir": 12,
            "fabrication": null,
            "adherence": 85,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 69,
            "nToolTurns": 9,
            "nCls": 58,
            "kStall": 53,
            "kDeadAir": 8,
            "kToolSilent": 8,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 3,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "hi": {
        "label": "Hindi",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 92.98,
            "taskScore": 93,
            "stalled": 13,
            "toolSilent": 72,
            "deadAir": 43,
            "fabrication": null,
            "adherence": 74,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 152,
            "nToolTurns": 92,
            "nCls": 86,
            "kStall": 7,
            "kDeadAir": 66,
            "kToolSilent": 66,
            "kAdherence": 86,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 90.35,
            "taskScore": 90,
            "stalled": 17,
            "toolSilent": 96,
            "deadAir": 58,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 153,
            "nToolTurns": 93,
            "nCls": 64,
            "kStall": 9,
            "kDeadAir": 89,
            "kToolSilent": 89,
            "kAdherence": 64,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 90.35,
            "taskScore": 90,
            "stalled": 13,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 188,
            "nToolTurns": 129,
            "nCls": 59,
            "kStall": 7,
            "kDeadAir": 129,
            "kToolSilent": 129,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 89.18,
            "taskScore": 89,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 65,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 171,
            "nToolTurns": 111,
            "nCls": 60,
            "kStall": 9,
            "kDeadAir": 111,
            "kToolSilent": 111,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 84.94,
            "taskScore": 85,
            "stalled": 11,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 184,
            "nToolTurns": 127,
            "nCls": 57,
            "kStall": 6,
            "kDeadAir": 127,
            "kToolSilent": 127,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 84.94,
            "taskScore": 85,
            "stalled": 19,
            "toolSilent": 87,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 159,
            "nToolTurns": 100,
            "nCls": 72,
            "kStall": 10,
            "kDeadAir": 87,
            "kToolSilent": 87,
            "kAdherence": 69,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 82.46,
            "taskScore": 82,
            "stalled": 26,
            "toolSilent": 81,
            "deadAir": 53,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 174,
            "nToolTurns": 114,
            "nCls": 82,
            "kStall": 14,
            "kDeadAir": 92,
            "kToolSilent": 92,
            "kAdherence": 82,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 81.14,
            "taskScore": 81,
            "stalled": 30,
            "toolSilent": 98,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 89,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 200,
            "nToolTurns": 144,
            "nCls": 59,
            "kStall": 16,
            "kDeadAir": 141,
            "kToolSilent": 141,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 80.85,
            "taskScore": 81,
            "stalled": 31,
            "toolSilent": 100,
            "deadAir": 63,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 159,
            "nToolTurns": 100,
            "nCls": 59,
            "kStall": 17,
            "kDeadAir": 100,
            "kToolSilent": 100,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 79.97,
            "taskScore": 80,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 181,
            "nToolTurns": 126,
            "nCls": 55,
            "kStall": 8,
            "kDeadAir": 126,
            "kToolSilent": 126,
            "kAdherence": 55,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 78.51,
            "taskScore": 79,
            "stalled": 35,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 73,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 142,
            "nToolTurns": 82,
            "nCls": 142,
            "kStall": 19,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 142,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 76.17,
            "taskScore": 76,
            "stalled": 39,
            "toolSilent": 1,
            "deadAir": 1,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 153,
            "nToolTurns": 93,
            "nCls": 152,
            "kStall": 21,
            "kDeadAir": 1,
            "kToolSilent": 1,
            "kAdherence": 152,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 75.73,
            "taskScore": 76,
            "stalled": 22,
            "toolSilent": 100,
            "deadAir": 72,
            "fabrication": null,
            "adherence": 96,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 181,
            "nToolTurns": 130,
            "nCls": 51,
            "kStall": 12,
            "kDeadAir": 130,
            "kToolSilent": 130,
            "kAdherence": 51,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 75.15,
            "taskScore": 75,
            "stalled": 33,
            "toolSilent": 22,
            "deadAir": 14,
            "fabrication": null,
            "adherence": 11,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 168,
            "nToolTurns": 108,
            "nCls": 144,
            "kStall": 18,
            "kDeadAir": 24,
            "kToolSilent": 24,
            "kAdherence": 135,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 74.27,
            "taskScore": 74,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 93,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 173,
            "nToolTurns": 119,
            "nCls": 54,
            "kStall": 13,
            "kDeadAir": 119,
            "kToolSilent": 119,
            "kAdherence": 54,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 73.25,
            "taskScore": 73,
            "stalled": 39,
            "toolSilent": 91,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 16,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 197,
            "nToolTurns": 143,
            "nCls": 67,
            "kStall": 21,
            "kDeadAir": 130,
            "kToolSilent": 130,
            "kAdherence": 62,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 72.81,
            "taskScore": 73,
            "stalled": 50,
            "toolSilent": 99,
            "deadAir": 52,
            "fabrication": null,
            "adherence": 57,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 127,
            "nToolTurns": 67,
            "nCls": 61,
            "kStall": 27,
            "kDeadAir": 66,
            "kToolSilent": 66,
            "kAdherence": 61,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 64.91,
            "taskScore": 65,
            "stalled": 67,
            "toolSilent": 100,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 20,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 134,
            "nToolTurns": 74,
            "nCls": 60,
            "kStall": 36,
            "kDeadAir": 74,
            "kToolSilent": 74,
            "kAdherence": 35,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 40.35,
            "taskScore": 40,
            "stalled": 28,
            "toolSilent": 100,
            "deadAir": 83,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 208,
            "nToolTurns": 172,
            "nCls": 36,
            "kStall": 15,
            "kDeadAir": 172,
            "kToolSilent": 172,
            "kAdherence": 36,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 34.65,
            "taskScore": 35,
            "stalled": 74,
            "toolSilent": 61,
            "deadAir": 19,
            "fabrication": null,
            "adherence": 86,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 88,
            "nToolTurns": 28,
            "nCls": 71,
            "kStall": 40,
            "kDeadAir": 17,
            "kToolSilent": 17,
            "kAdherence": 71,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 10,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 30.99,
            "taskScore": 31,
            "stalled": 85,
            "toolSilent": 100,
            "deadAir": 41,
            "fabrication": null,
            "adherence": 53,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 101,
            "nToolTurns": 41,
            "nCls": 60,
            "kStall": 46,
            "kDeadAir": 41,
            "kToolSilent": 41,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 15,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "ta": {
        "label": "Tamil",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 93.86,
            "taskScore": 94,
            "stalled": 7,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 193,
            "nToolTurns": 135,
            "nCls": 58,
            "kStall": 4,
            "kDeadAir": 135,
            "kToolSilent": 135,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 92.98,
            "taskScore": 93,
            "stalled": 9,
            "toolSilent": 92,
            "deadAir": 59,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 167,
            "nToolTurns": 107,
            "nCls": 69,
            "kStall": 5,
            "kDeadAir": 98,
            "kToolSilent": 98,
            "kAdherence": 68,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 88.16,
            "taskScore": 88,
            "stalled": 19,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 177,
            "nToolTurns": 117,
            "nCls": 60,
            "kStall": 10,
            "kDeadAir": 117,
            "kToolSilent": 117,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 87.13,
            "taskScore": 87,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 61,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 152,
            "nToolTurns": 92,
            "nCls": 60,
            "kStall": 11,
            "kDeadAir": 92,
            "kToolSilent": 92,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 85.67,
            "taskScore": 86,
            "stalled": 33,
            "toolSilent": 83,
            "deadAir": 47,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 137,
            "nToolTurns": 77,
            "nCls": 73,
            "kStall": 18,
            "kDeadAir": 64,
            "kToolSilent": 64,
            "kAdherence": 73,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 83.33,
            "taskScore": 83,
            "stalled": 20,
            "toolSilent": 85,
            "deadAir": 58,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 173,
            "nToolTurns": 114,
            "nCls": 72,
            "kStall": 11,
            "kDeadAir": 101,
            "kToolSilent": 97,
            "kAdherence": 72,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 82.02,
            "taskScore": 82,
            "stalled": 28,
            "toolSilent": 99,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 178,
            "nToolTurns": 121,
            "nCls": 58,
            "kStall": 15,
            "kDeadAir": 120,
            "kToolSilent": 120,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 81.88,
            "taskScore": 82,
            "stalled": 14,
            "toolSilent": 100,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 0,
            "nTask": 46,
            "nFab": 0,
            "nStall": 44,
            "nTurns": 156,
            "nToolTurns": 110,
            "nCls": 46,
            "kStall": 6,
            "kDeadAir": 110,
            "kToolSilent": 110,
            "kAdherence": 46,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 81.73,
            "taskScore": 82,
            "stalled": 26,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 13,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 152,
            "nToolTurns": 92,
            "nCls": 152,
            "kStall": 14,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 152,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 79.97,
            "taskScore": 80,
            "stalled": 20,
            "toolSilent": 99,
            "deadAir": 65,
            "fabrication": null,
            "adherence": 83,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 166,
            "nToolTurns": 109,
            "nCls": 58,
            "kStall": 11,
            "kDeadAir": 108,
            "kToolSilent": 108,
            "kAdherence": 55,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 77.19,
            "taskScore": 77,
            "stalled": 44,
            "toolSilent": 96,
            "deadAir": 52,
            "fabrication": null,
            "adherence": 64,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 132,
            "nToolTurns": 72,
            "nCls": 63,
            "kStall": 24,
            "kDeadAir": 69,
            "kToolSilent": 69,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 76.64,
            "taskScore": 77,
            "stalled": 28,
            "toolSilent": 34,
            "deadAir": 23,
            "fabrication": null,
            "adherence": 4,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 179,
            "nToolTurns": 122,
            "nCls": 137,
            "kStall": 15,
            "kDeadAir": 42,
            "kToolSilent": 41,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 76.32,
            "taskScore": 76,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 178,
            "nToolTurns": 122,
            "nCls": 56,
            "kStall": 13,
            "kDeadAir": 122,
            "kToolSilent": 122,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 75.88,
            "taskScore": 76,
            "stalled": 28,
            "toolSilent": 96,
            "deadAir": 72,
            "fabrication": null,
            "adherence": 75,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 208,
            "nToolTurns": 156,
            "nCls": 59,
            "kStall": 15,
            "kDeadAir": 149,
            "kToolSilent": 149,
            "kAdherence": 55,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 74.56,
            "taskScore": 75,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 72,
            "fabrication": null,
            "adherence": 80,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 180,
            "nToolTurns": 130,
            "nCls": 50,
            "kStall": 11,
            "kDeadAir": 130,
            "kToolSilent": 130,
            "kAdherence": 50,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 73.66,
            "taskScore": 74,
            "stalled": 49,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 100,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 131,
            "nToolTurns": 72,
            "nCls": 131,
            "kStall": 26,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 131,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 62.72,
            "taskScore": 63,
            "stalled": 50,
            "toolSilent": 89,
            "deadAir": 64,
            "fabrication": null,
            "adherence": 3,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 191,
            "nToolTurns": 138,
            "nCls": 68,
            "kStall": 27,
            "kDeadAir": 123,
            "kToolSilent": 123,
            "kAdherence": 64,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 54.82,
            "taskScore": 55,
            "stalled": 78,
            "toolSilent": 95,
            "deadAir": 48,
            "fabrication": null,
            "adherence": 25,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 120,
            "nToolTurns": 60,
            "nCls": 63,
            "kStall": 42,
            "kDeadAir": 57,
            "kToolSilent": 57,
            "kAdherence": 35,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 31.58,
            "taskScore": 32,
            "stalled": 30,
            "toolSilent": 100,
            "deadAir": 85,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 226,
            "nToolTurns": 192,
            "nCls": 34,
            "kStall": 16,
            "kDeadAir": 192,
            "kToolSilent": 192,
            "kAdherence": 34,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 30.99,
            "taskScore": 31,
            "stalled": 78,
            "toolSilent": 41,
            "deadAir": 13,
            "fabrication": null,
            "adherence": 65,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 87,
            "nToolTurns": 27,
            "nCls": 76,
            "kStall": 42,
            "kDeadAir": 11,
            "kToolSilent": 11,
            "kAdherence": 76,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 12,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 21.49,
            "taskScore": 21,
            "stalled": 96,
            "toolSilent": 100,
            "deadAir": 37,
            "fabrication": null,
            "adherence": 77,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 95,
            "nToolTurns": 35,
            "nCls": 60,
            "kStall": 52,
            "kDeadAir": 35,
            "kToolSilent": 35,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 14,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "te": {
        "label": "Telugu",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 93.57,
            "taskScore": 94,
            "stalled": 9,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 174,
            "nToolTurns": 115,
            "nCls": 59,
            "kStall": 5,
            "kDeadAir": 115,
            "kToolSilent": 115,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 91.23,
            "taskScore": 91,
            "stalled": 13,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 189,
            "nToolTurns": 129,
            "nCls": 60,
            "kStall": 7,
            "kDeadAir": 129,
            "kToolSilent": 129,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 88.6,
            "taskScore": 89,
            "stalled": 20,
            "toolSilent": 99,
            "deadAir": 59,
            "fabrication": null,
            "adherence": 95,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 150,
            "nToolTurns": 90,
            "nCls": 61,
            "kStall": 11,
            "kDeadAir": 89,
            "kToolSilent": 89,
            "kAdherence": 61,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 87.87,
            "taskScore": 88,
            "stalled": 11,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 181,
            "nToolTurns": 125,
            "nCls": 56,
            "kStall": 6,
            "kDeadAir": 125,
            "kToolSilent": 125,
            "kAdherence": 56,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 87.13,
            "taskScore": 87,
            "stalled": 9,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 179,
            "nToolTurns": 122,
            "nCls": 57,
            "kStall": 5,
            "kDeadAir": 122,
            "kToolSilent": 122,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 85.67,
            "taskScore": 86,
            "stalled": 22,
            "toolSilent": 95,
            "deadAir": 59,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 158,
            "nToolTurns": 98,
            "nCls": 65,
            "kStall": 12,
            "kDeadAir": 93,
            "kToolSilent": 93,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 85.38,
            "taskScore": 85,
            "stalled": 20,
            "toolSilent": 98,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 173,
            "nToolTurns": 116,
            "nCls": 59,
            "kStall": 11,
            "kDeadAir": 114,
            "kToolSilent": 114,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 82.16,
            "taskScore": 82,
            "stalled": 35,
            "toolSilent": 81,
            "deadAir": 45,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 137,
            "nToolTurns": 77,
            "nCls": 75,
            "kStall": 19,
            "kDeadAir": 62,
            "kToolSilent": 62,
            "kAdherence": 75,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 81.73,
            "taskScore": 82,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 87,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 174,
            "nToolTurns": 120,
            "nCls": 54,
            "kStall": 11,
            "kDeadAir": 120,
            "kToolSilent": 120,
            "kAdherence": 52,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 81.58,
            "taskScore": 82,
            "stalled": 35,
            "toolSilent": 100,
            "deadAir": 56,
            "fabrication": null,
            "adherence": 54,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 135,
            "nToolTurns": 75,
            "nCls": 60,
            "kStall": 19,
            "kDeadAir": 75,
            "kToolSilent": 75,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 81.29,
            "taskScore": 81,
            "stalled": 26,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 11,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 150,
            "nToolTurns": 90,
            "nCls": 150,
            "kStall": 14,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 149,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 79.53,
            "taskScore": 80,
            "stalled": 24,
            "toolSilent": 77,
            "deadAir": 51,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 171,
            "nToolTurns": 111,
            "nCls": 84,
            "kStall": 13,
            "kDeadAir": 87,
            "kToolSilent": 85,
            "kAdherence": 84,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 78.51,
            "taskScore": 79,
            "stalled": 33,
            "toolSilent": 94,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 50,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 194,
            "nToolTurns": 141,
            "nCls": 61,
            "kStall": 18,
            "kDeadAir": 133,
            "kToolSilent": 133,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 78.22,
            "taskScore": 78,
            "stalled": 33,
            "toolSilent": 17,
            "deadAir": 11,
            "fabrication": null,
            "adherence": 8,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 167,
            "nToolTurns": 109,
            "nCls": 148,
            "kStall": 18,
            "kDeadAir": 19,
            "kToolSilent": 19,
            "kAdherence": 126,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 76.9,
            "taskScore": 77,
            "stalled": 15,
            "toolSilent": 100,
            "deadAir": 77,
            "fabrication": null,
            "adherence": 60,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 200,
            "nToolTurns": 153,
            "nCls": 47,
            "kStall": 8,
            "kDeadAir": 153,
            "kToolSilent": 153,
            "kAdherence": 47,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 66.96,
            "taskScore": 67,
            "stalled": 57,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 132,
            "nToolTurns": 72,
            "nCls": 132,
            "kStall": 31,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 132,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 61.7,
            "taskScore": 62,
            "stalled": 39,
            "toolSilent": 87,
            "deadAir": 65,
            "fabrication": null,
            "adherence": 1,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 203,
            "nToolTurns": 151,
            "nCls": 70,
            "kStall": 21,
            "kDeadAir": 132,
            "kToolSilent": 132,
            "kAdherence": 15,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 59.21,
            "taskScore": 59,
            "stalled": 69,
            "toolSilent": 98,
            "deadAir": 51,
            "fabrication": null,
            "adherence": 9,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 125,
            "nToolTurns": 65,
            "nCls": 61,
            "kStall": 37,
            "kDeadAir": 64,
            "kToolSilent": 64,
            "kAdherence": 49,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 43.86,
            "taskScore": 44,
            "stalled": 28,
            "toolSilent": 100,
            "deadAir": 83,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 229,
            "nToolTurns": 191,
            "nCls": 38,
            "kStall": 15,
            "kDeadAir": 191,
            "kToolSilent": 191,
            "kAdherence": 38,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 22.22,
            "taskScore": 22,
            "stalled": 94,
            "toolSilent": 100,
            "deadAir": 39,
            "fabrication": null,
            "adherence": 80,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 99,
            "nToolTurns": 39,
            "nCls": 60,
            "kStall": 51,
            "kDeadAir": 39,
            "kToolSilent": 39,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 14,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 6.58,
            "taskScore": 7,
            "stalled": 94,
            "toolSilent": 67,
            "deadAir": 6,
            "fabrication": null,
            "adherence": 80,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 66,
            "nToolTurns": 6,
            "nCls": 62,
            "kStall": 51,
            "kDeadAir": 4,
            "kToolSilent": 4,
            "kAdherence": 62,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 2,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "zh": {
        "label": "Chinese",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-23",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 90.35,
            "taskScore": 90,
            "stalled": 13,
            "toolSilent": 100,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 184,
            "nToolTurns": 124,
            "nCls": 60,
            "kStall": 7,
            "kDeadAir": 124,
            "kToolSilent": 124,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 89.47,
            "taskScore": 89,
            "stalled": 19,
            "toolSilent": 70,
            "deadAir": 42,
            "fabrication": null,
            "adherence": 8,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 150,
            "nToolTurns": 90,
            "nCls": 87,
            "kStall": 10,
            "kDeadAir": 63,
            "kToolSilent": 63,
            "kAdherence": 87,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 89.33,
            "taskScore": 89,
            "stalled": 17,
            "toolSilent": 93,
            "deadAir": 57,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 157,
            "nToolTurns": 97,
            "nCls": 67,
            "kStall": 9,
            "kDeadAir": 90,
            "kToolSilent": 90,
            "kAdherence": 67,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 88.6,
            "taskScore": 89,
            "stalled": 20,
            "toolSilent": 99,
            "deadAir": 62,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 159,
            "nToolTurns": 99,
            "nCls": 61,
            "kStall": 11,
            "kDeadAir": 98,
            "kToolSilent": 98,
            "kAdherence": 61,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 87.43,
            "taskScore": 87,
            "stalled": 20,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 173,
            "nToolTurns": 114,
            "nCls": 59,
            "kStall": 11,
            "kDeadAir": 114,
            "kToolSilent": 114,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 85.96,
            "taskScore": 86,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 177,
            "nToolTurns": 120,
            "nCls": 57,
            "kStall": 9,
            "kDeadAir": 120,
            "kToolSilent": 120,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 84.06,
            "taskScore": 84,
            "stalled": 22,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 172,
            "nToolTurns": 114,
            "nCls": 58,
            "kStall": 12,
            "kDeadAir": 114,
            "kToolSilent": 114,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 81.58,
            "taskScore": 82,
            "stalled": 19,
            "toolSilent": 100,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 180,
            "nToolTurns": 123,
            "nCls": 57,
            "kStall": 10,
            "kDeadAir": 123,
            "kToolSilent": 123,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 81.14,
            "taskScore": 81,
            "stalled": 22,
            "toolSilent": 7,
            "deadAir": 5,
            "fabrication": null,
            "adherence": 2,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 177,
            "nToolTurns": 121,
            "nCls": 168,
            "kStall": 12,
            "kDeadAir": 9,
            "kToolSilent": 9,
            "kAdherence": 135,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 79.82,
            "taskScore": 80,
            "stalled": 31,
            "toolSilent": 3,
            "deadAir": 2,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 158,
            "nToolTurns": 99,
            "nCls": 155,
            "kStall": 17,
            "kDeadAir": 3,
            "kToolSilent": 3,
            "kAdherence": 155,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 78.07,
            "taskScore": 78,
            "stalled": 26,
            "toolSilent": 100,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 65,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 184,
            "nToolTurns": 130,
            "nCls": 54,
            "kStall": 14,
            "kDeadAir": 130,
            "kToolSilent": 130,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 73.96,
            "taskScore": 74,
            "stalled": 34,
            "toolSilent": 92,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 97,
            "nTask": 56,
            "nFab": 0,
            "nStall": 53,
            "nTurns": 197,
            "nToolTurns": 146,
            "nCls": 62,
            "kStall": 18,
            "kDeadAir": 135,
            "kToolSilent": 135,
            "kAdherence": 61,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 73.68,
            "taskScore": 74,
            "stalled": 31,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 86,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 180,
            "nToolTurns": 126,
            "nCls": 54,
            "kStall": 17,
            "kDeadAir": 126,
            "kToolSilent": 126,
            "kAdherence": 54,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 73.39,
            "taskScore": 73,
            "stalled": 28,
            "toolSilent": 62,
            "deadAir": 42,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 163,
            "nToolTurns": 103,
            "nCls": 94,
            "kStall": 15,
            "kDeadAir": 69,
            "kToolSilent": 64,
            "kAdherence": 94,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 72.81,
            "taskScore": 73,
            "stalled": 48,
            "toolSilent": 100,
            "deadAir": 53,
            "fabrication": null,
            "adherence": 35,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 127,
            "nToolTurns": 67,
            "nCls": 60,
            "kStall": 26,
            "kDeadAir": 67,
            "kToolSilent": 67,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 72.66,
            "taskScore": 73,
            "stalled": 39,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 87,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 149,
            "nToolTurns": 89,
            "nCls": 149,
            "kStall": 21,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 149,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 64.47,
            "taskScore": 64,
            "stalled": 67,
            "toolSilent": 100,
            "deadAir": 55,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 133,
            "nToolTurns": 73,
            "nCls": 60,
            "kStall": 36,
            "kDeadAir": 73,
            "kToolSilent": 73,
            "kAdherence": 40,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 61.84,
            "taskScore": 62,
            "stalled": 44,
            "toolSilent": 91,
            "deadAir": 68,
            "fabrication": null,
            "adherence": 77,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 201,
            "nToolTurns": 150,
            "nCls": 64,
            "kStall": 24,
            "kDeadAir": 137,
            "kToolSilent": 137,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 29.82,
            "taskScore": 30,
            "stalled": 0,
            "toolSilent": 100,
            "deadAir": 99,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 273,
            "nToolTurns": 269,
            "nCls": 4,
            "kStall": 0,
            "kDeadAir": 269,
            "kToolSilent": 269,
            "kAdherence": 4,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 25,
            "taskScore": 25,
            "stalled": 93,
            "toolSilent": 100,
            "deadAir": 33,
            "fabrication": null,
            "adherence": 88,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 90,
            "nToolTurns": 30,
            "nCls": 60,
            "kStall": 50,
            "kDeadAir": 30,
            "kToolSilent": 30,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 15,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 23.25,
            "taskScore": 23,
            "stalled": 81,
            "toolSilent": 86,
            "deadAir": 22,
            "fabrication": null,
            "adherence": 91,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 81,
            "nToolTurns": 21,
            "nCls": 63,
            "kStall": 44,
            "kDeadAir": 18,
            "kToolSilent": 18,
            "kAdherence": 63,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 9,
            "iCls": 19,
            "iFab": 0
          }
        ]
      },
      "ko": {
        "label": "Korean",
        "meta": "same scenarios + v3 grader as the English board · n=57 task / 0 fab · 2026-08-26",
        "rows": [
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "taskDone": 87.72,
            "taskScore": 88,
            "stalled": 9,
            "toolSilent": 100,
            "deadAir": 69,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 185,
            "nToolTurns": 128,
            "nCls": 57,
            "kStall": 5,
            "kDeadAir": 128,
            "kToolSilent": 128,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "taskDone": 87.72,
            "taskScore": 88,
            "stalled": 22,
            "toolSilent": 100,
            "deadAir": 63,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 162,
            "nToolTurns": 102,
            "nCls": 60,
            "kStall": 12,
            "kDeadAir": 102,
            "kToolSilent": 102,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "taskDone": 87.72,
            "taskScore": 88,
            "stalled": 28,
            "toolSilent": 75,
            "deadAir": 45,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 148,
            "nToolTurns": 88,
            "nCls": 82,
            "kStall": 15,
            "kDeadAir": 66,
            "kToolSilent": 66,
            "kAdherence": 82,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "taskDone": 87.72,
            "taskScore": 88,
            "stalled": 22,
            "toolSilent": 100,
            "deadAir": 64,
            "fabrication": null,
            "adherence": 93,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 169,
            "nToolTurns": 109,
            "nCls": 60,
            "kStall": 12,
            "kDeadAir": 109,
            "kToolSilent": 109,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-terra",
            "taskDone": 86.84,
            "taskScore": 87,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 93,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 174,
            "nToolTurns": 114,
            "nCls": 60,
            "kStall": 13,
            "kDeadAir": 114,
            "kToolSilent": 114,
            "kAdherence": 59,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "taskDone": 85.67,
            "taskScore": 86,
            "stalled": 26,
            "toolSilent": 95,
            "deadAir": 58,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 156,
            "nToolTurns": 96,
            "nCls": 65,
            "kStall": 14,
            "kDeadAir": 91,
            "kToolSilent": 91,
            "kAdherence": 64,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5.6-luna",
            "taskDone": 83.33,
            "taskScore": 83,
            "stalled": 24,
            "toolSilent": 100,
            "deadAir": 71,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 198,
            "nToolTurns": 140,
            "nCls": 58,
            "kStall": 13,
            "kDeadAir": 140,
            "kToolSilent": 140,
            "kAdherence": 58,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Haiku 4.5",
            "taskDone": 81.29,
            "taskScore": 81,
            "stalled": 33,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 65,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 157,
            "nToolTurns": 97,
            "nCls": 157,
            "kStall": 18,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 157,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "gpt-oss-120b",
            "taskDone": 80.7,
            "taskScore": 81,
            "stalled": 17,
            "toolSilent": 100,
            "deadAir": 70,
            "fabrication": null,
            "adherence": 0,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 187,
            "nToolTurns": 130,
            "nCls": 57,
            "kStall": 9,
            "kDeadAir": 130,
            "kToolSilent": 130,
            "kAdherence": 57,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Anthropic",
            "model": "Claude Sonnet 5",
            "taskDone": 78.36,
            "taskScore": 78,
            "stalled": 30,
            "toolSilent": 53,
            "deadAir": 34,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 170,
            "nToolTurns": 110,
            "nCls": 112,
            "kStall": 16,
            "kDeadAir": 58,
            "kToolSilent": 58,
            "kAdherence": 112,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 18,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling",
            "taskDone": 76.75,
            "taskScore": 77,
            "stalled": 31,
            "toolSilent": 89,
            "deadAir": 66,
            "fabrication": null,
            "adherence": 93,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 206,
            "nToolTurns": 151,
            "nCls": 71,
            "kStall": 17,
            "kDeadAir": 135,
            "kToolSilent": 135,
            "kAdherence": 70,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "DeepSeek-V4-Flash-0731",
            "taskDone": 73.1,
            "taskScore": 73,
            "stalled": 35,
            "toolSilent": 21,
            "deadAir": 14,
            "fabrication": null,
            "adherence": 1,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 181,
            "nToolTurns": 125,
            "nCls": 155,
            "kStall": 19,
            "kDeadAir": 26,
            "kToolSilent": 26,
            "kAdherence": 79,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Gemini",
            "model": "gemini-3.5-flash-lite",
            "taskDone": 71.78,
            "taskScore": 72,
            "stalled": 31,
            "toolSilent": 100,
            "deadAir": 67,
            "fabrication": null,
            "adherence": 98,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 171,
            "nToolTurns": 115,
            "nCls": 56,
            "kStall": 17,
            "kDeadAir": 115,
            "kToolSilent": 115,
            "kAdherence": 54,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "Nemotron-3-Ultra",
            "taskDone": 67.98,
            "taskScore": 68,
            "stalled": 57,
            "toolSilent": 98,
            "deadAir": 56,
            "fabrication": null,
            "adherence": 55,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 140,
            "nToolTurns": 80,
            "nCls": 62,
            "kStall": 31,
            "kDeadAir": 78,
            "kToolSilent": 78,
            "kAdherence": 53,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "taskDone": 67.69,
            "taskScore": 68,
            "stalled": 35,
            "toolSilent": 100,
            "deadAir": 72,
            "fabrication": null,
            "adherence": 92,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 176,
            "nToolTurns": 126,
            "nCls": 50,
            "kStall": 19,
            "kDeadAir": 126,
            "kToolSilent": 126,
            "kAdherence": 50,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "inkling-small",
            "taskDone": 65.79,
            "taskScore": 66,
            "stalled": 37,
            "toolSilent": 94,
            "deadAir": 73,
            "fabrication": null,
            "adherence": 59,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 211,
            "nToolTurns": 165,
            "nCls": 56,
            "kStall": 20,
            "kDeadAir": 155,
            "kToolSilent": 155,
            "kAdherence": 55,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "taskDone": 65.79,
            "taskScore": 66,
            "stalled": 57,
            "toolSilent": 92,
            "deadAir": 46,
            "fabrication": null,
            "adherence": 92,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 120,
            "nToolTurns": 60,
            "nCls": 65,
            "kStall": 31,
            "kDeadAir": 55,
            "kToolSilent": 55,
            "kAdherence": 65,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Baseten",
            "model": "GLM-4.7",
            "taskDone": 64.62,
            "taskScore": 65,
            "stalled": 61,
            "toolSilent": 0,
            "deadAir": 0,
            "fabrication": null,
            "adherence": 100,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 146,
            "nToolTurns": 86,
            "nCls": 146,
            "kStall": 33,
            "kDeadAir": 0,
            "kToolSilent": 0,
            "kAdherence": 146,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "taskDone": 32.46,
            "taskScore": 32,
            "stalled": 0,
            "toolSilent": 100,
            "deadAir": 100,
            "fabrication": null,
            "adherence": 6,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 285,
            "nToolTurns": 285,
            "nCls": 0,
            "kStall": 0,
            "kDeadAir": 285,
            "kToolSilent": 285,
            "kAdherence": 0,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 19,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-nano",
            "taskDone": 17.98,
            "taskScore": 18,
            "stalled": 93,
            "toolSilent": 100,
            "deadAir": 28,
            "fabrication": null,
            "adherence": 97,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 83,
            "nToolTurns": 23,
            "nCls": 60,
            "kStall": 50,
            "kDeadAir": 23,
            "kToolSilent": 23,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 15,
            "iCls": 19,
            "iFab": 0
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "taskDone": 13.6,
            "taskScore": 14,
            "stalled": 94,
            "toolSilent": 100,
            "deadAir": 26,
            "fabrication": null,
            "adherence": 70,
            "nTask": 57,
            "nFab": 0,
            "nStall": 54,
            "nTurns": 81,
            "nToolTurns": 21,
            "nCls": 60,
            "kStall": 51,
            "kDeadAir": 21,
            "kToolSilent": 21,
            "kAdherence": 60,
            "kFab": 0,
            "iStall": 18,
            "iTurns": 19,
            "iToolTurns": 7,
            "iCls": 19,
            "iFab": 0
          }
        ]
      }
    },
    "llmFidelity": {
      "en": {
        "label": "English",
        "meta": "control column · n=21 values / model · 2026-07-28",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "id": "openai:gpt-4.1-mini",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "id": "openai:gpt-5",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "id": "openai:gpt-5-mini",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "id": "cerebras:gpt-oss-120b",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "id": "cerebras:gemma-4-31b",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Together",
            "model": "DeepSeek-V4-Pro",
            "id": "together:deepseek-ai/DeepSeek-V4-Pro",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "id": "together:meta-llama/Llama-3.3-70B-Instruct-Turbo",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "id": "xai:grok-4.3",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "id": "openai:gpt-4.1",
            "fidelity": 95,
            "tone": "good"
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "id": "alibaba:qwen-turbo",
            "fidelity": 86,
            "tone": "warn",
            "flag": "Gives the appointment date with no year."
          }
        ]
      },
      "es": {
        "label": "Spanish",
        "meta": "n=21 values / model · 2026-07-28",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "id": "openai:gpt-4.1",
            "fidelity": 100,
            "band": [
              94,
              100
            ],
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "id": "openai:gpt-4.1-mini",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "id": "openai:gpt-5",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "id": "openai:gpt-5-mini",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "id": "cerebras:gpt-oss-120b",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "id": "cerebras:gemma-4-31b",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "id": "together:meta-llama/Llama-3.3-70B-Instruct-Turbo",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "id": "xai:grok-4.3",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Together",
            "model": "DeepSeek-V4-Pro",
            "id": "together:deepseek-ai/DeepSeek-V4-Pro",
            "fidelity": 90,
            "band": [
              90,
              100
            ],
            "tone": "warn"
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "id": "alibaba:qwen-turbo",
            "fidelity": 86,
            "tone": "warn"
          }
        ]
      },
      "ar": {
        "label": "Arabic",
        "meta": "n=21 values / model · 2026-07-28",
        "rows": [
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "id": "cerebras:gemma-4-31b",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "id": "xai:grok-4.3",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "id": "openai:gpt-4.1",
            "fidelity": 95,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "id": "openai:gpt-5",
            "fidelity": 90,
            "band": [
              90,
              95
            ],
            "tone": "warn"
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "id": "together:meta-llama/Llama-3.3-70B-Instruct-Turbo",
            "fidelity": 90,
            "band": [
              86,
              90
            ],
            "tone": "warn",
            "flag": "Read a 14:45 appointment back as 12:45."
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "id": "cerebras:gpt-oss-120b",
            "fidelity": 81,
            "band": [
              81,
              86
            ],
            "tone": "warn",
            "flag": "Said $57.50 for a $75.50 total, and 2026 for a 2027 date."
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "id": "openai:gpt-4.1-mini",
            "fidelity": 81,
            "band": [
              76,
              81
            ],
            "tone": "warn",
            "flag": "Said 2007 for a 2027 date on every run; also $57.50 for $75.50."
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "id": "openai:gpt-5-mini",
            "fidelity": 71,
            "tone": "warn",
            "flag": "Said $57.50 for a $75.50 total; emitted Chinese numerals mid-sentence for the cents."
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "id": "alibaba:qwen-turbo",
            "fidelity": 43,
            "tone": "bad",
            "flag": "Said $70.45 for a $75.50 total on every run."
          }
        ]
      },
      "fil": {
        "label": "Filipino",
        "meta": "n=21 values / model · 2026-07-28",
        "rows": [
          {
            "provider": "OpenAI",
            "model": "gpt-4.1",
            "id": "openai:gpt-4.1",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5",
            "id": "openai:gpt-5",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "xAI",
            "model": "Grok-4.3",
            "id": "xai:grok-4.3",
            "fidelity": 100,
            "tone": "good"
          },
          {
            "provider": "Together",
            "model": "Llama-3.3-70B",
            "id": "together:meta-llama/Llama-3.3-70B-Instruct-Turbo",
            "fidelity": 100,
            "band": [
              95,
              100
            ],
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-4.1-mini",
            "id": "openai:gpt-4.1-mini",
            "fidelity": 95,
            "tone": "good"
          },
          {
            "provider": "Cerebras",
            "model": "gemma-4-31b",
            "id": "cerebras:gemma-4-31b",
            "fidelity": 95,
            "tone": "good"
          },
          {
            "provider": "OpenAI",
            "model": "gpt-5-mini",
            "id": "openai:gpt-5-mini",
            "fidelity": 90,
            "tone": "warn"
          },
          {
            "provider": "Cerebras",
            "model": "gpt-oss-120b",
            "id": "cerebras:gpt-oss-120b",
            "fidelity": 86,
            "tone": "warn"
          },
          {
            "provider": "Alibaba",
            "model": "qwen-turbo",
            "id": "alibaba:qwen-turbo",
            "fidelity": 67,
            "band": [
              52,
              67
            ],
            "tone": "bad",
            "flag": "Invented a November delivery date for an August order."
          }
        ]
      }
    }
  },
  "recommended": [
    {
      "n": "01",
      "title": "Real-time phone agent",
      "legs": [
        {
          "mod": "STT",
          "pick": "AssemblyAI Universal-3.5 Pro",
          "why": "66ms"
        },
        {
          "mod": "LLM",
          "pick": "Cerebras gemma-4-31b",
          "why": "192ms"
        },
        {
          "mod": "TTS",
          "pick": "Inworld inworld-tts-2",
          "why": "116ms"
        }
      ],
      "note": "Lowest-latency legs on the path a live call actually runs. Palabra reaches first audio sooner but drops minus signs, so it does not get the voice slot."
    },
    {
      "n": "02",
      "title": "Accuracy-critical",
      "legs": [
        {
          "mod": "STT",
          "pick": "AssemblyAI Universal-3.5 Pro",
          "why": "2.0% WER"
        },
        {
          "mod": "LLM",
          "pick": "Anthropic Claude Haiku 4.5",
          "why": "0% fabrication"
        },
        {
          "mod": "TTS",
          "pick": "Inworld inworld-tts-2",
          "why": "0.93 robustness"
        }
      ],
      "note": "Support & healthcare. Accuracy has to include the voice, and inworld-tts-2 leads the field by a wide margin on speaking the value it was handed."
    },
    {
      "n": "03",
      "title": "Natural conversation",
      "legs": [
        {
          "mod": "STT",
          "pick": "AssemblyAI Universal-3.5 Pro",
          "why": "2.0% WER"
        },
        {
          "mod": "LLM",
          "pick": "Anthropic Claude Haiku 4.5",
          "why": "1.6% dead-air"
        },
        {
          "mod": "TTS",
          "pick": "ElevenLabs eleven_v3_conversational",
          "why": "MOS 1590"
        }
      ],
      "note": "Companion, sales, brand voice. eleven_v3_conversational ties gemini-3.1-flash-tts at the top of naturalness; the tie breaks on time to first audio."
    },
    {
      "n": "04",
      "title": "Tool-heavy agent",
      "legs": [
        {
          "mod": "STT",
          "pick": "AssemblyAI Universal-3.5 Pro",
          "why": "2.0% WER"
        },
        {
          "mod": "LLM",
          "pick": "Anthropic Claude Haiku 4.5",
          "why": "2.7% tool silence"
        },
        {
          "mod": "TTS",
          "pick": "Inworld inworld-tts-2",
          "why": "116ms"
        }
      ],
      "note": "Scheduling & transactions. gpt-5-nano is cheaper still, but it is silent on every tool call and completes a third of the tasks."
    }
  ],
  "recommendedByLanguage": [
    {
      "code": "es",
      "label": "Spanish",
      "legs": [
        {
          "mod": "STT",
          "picks": [
            {
              "provider": "ElevenLabs",
              "model": "scribe_v2_realtime",
              "id": "elevenlabs:scribe_v2_realtime",
              "value": 3.8
            },
            {
              "provider": "Soniox",
              "model": "stt-rt-v5",
              "id": "soniox:stt-rt-v5",
              "value": 4.9
            }
          ],
          "unit": "% WER",
          "note": "English-Spanish code-switch. ElevenLabs 3.8% at 100% both-language coverage, Soniox 4.9% at 97%. This board publishes no confidence interval, so neither is shown ahead of the other. Monolingual Spanish is measured only on the batch path, which is not what a live agent runs on.",
          "caption": "EN-ES code-switch, no CI published"
        },
        {
          "mod": "LLM",
          "picks": [
            {
              "provider": "Anthropic",
              "model": "Claude Haiku 4.5",
              "id": "anthropic:claude-haiku-4-5",
              "value": 82.46
            }
          ],
          "unit": "%",
          "note": "The same recommendation on every language, carried across from our LLM board. Task completion is now measured per language: 21 models across ten languages, 19 scenarios at 3 iterations each, with a confidence interval on every cell — see the Multilingual LLM board. Nothing scores 100%; the English leader is gpt-5.6-luna at 94.21%, and the ten-language means run 68.8% to 74.0%, so language moves this axis far less than model choice does. Claude Haiku 4.5 is picked for behaviour rather than for topping the axis: the board publishes no composite score and no board-wide rank.",
          "caption": "measured in English"
        },
        {
          "mod": "TTS",
          "picks": [
            {
              "provider": "ElevenLabs",
              "model": "eleven_v3_conversational",
              "id": "elevenlabs:eleven_v3_conversational",
              "value": 1951
            }
          ],
          "unit": "Elo",
          "note": "Blind A/B listening study, arena Elo with 95% CI. The only language where one voice separates: ElevenLabs 1951 (95% CI 1859-2142) sits 200 Elo above Cartesia 1751 (1662-1858), but the intervals miss each other by a single point, so a refit could turn this into a tie."
        }
      ]
    },
    {
      "code": "de",
      "label": "German",
      "legs": [
        {
          "mod": "STT",
          "picks": [
            {
              "provider": "ElevenLabs",
              "model": "scribe_v2_realtime",
              "id": "elevenlabs:scribe_v2_realtime",
              "value": 3.1
            },
            {
              "provider": "Soniox",
              "model": "stt-rt-v5",
              "id": "soniox:stt-rt-v5",
              "value": 5.1
            }
          ],
          "unit": "% WER",
          "note": "English-German code-switch. ElevenLabs 3.1% at 98% both-language coverage, Soniox 5.1% at 100%. Soniox is last of 11 on the monolingual German batch board (13.4% WER) and the stronger of the two on coverage here: monolingual rank does not predict bilingual behaviour.",
          "caption": "EN-DE code-switch, no CI published"
        },
        {
          "mod": "LLM",
          "picks": [
            {
              "provider": "Anthropic",
              "model": "Claude Haiku 4.5",
              "id": "anthropic:claude-haiku-4-5",
              "value": 82.46
            }
          ],
          "unit": "%",
          "note": "The same recommendation on every language, carried across from our LLM board. Task completion is now measured per language: 21 models across ten languages, 19 scenarios at 3 iterations each, with a confidence interval on every cell — see the Multilingual LLM board. Nothing scores 100%; the English leader is gpt-5.6-luna at 94.21%, and the ten-language means run 68.8% to 74.0%, so language moves this axis far less than model choice does. Claude Haiku 4.5 is picked for behaviour rather than for topping the axis: the board publishes no composite score and no board-wide rank.",
          "caption": "measured in English"
        },
        {
          "mod": "TTS",
          "picks": [
            {
              "provider": "Cartesia",
              "model": "sonic-3.5",
              "id": "cartesia:sonic-3.5",
              "value": 1665
            },
            {
              "provider": "ElevenLabs",
              "model": "eleven_v3_conversational",
              "id": "elevenlabs:eleven_v3_conversational",
              "value": 1798
            },
            {
              "provider": "Soniox",
              "model": "tts-rt-v1",
              "id": "soniox:tts-rt-v1",
              "value": 1717
            },
            {
              "provider": "xAI Grok",
              "model": "grok-tts",
              "id": "xai:tts",
              "value": 1762
            }
          ],
          "unit": "Elo",
          "note": "Blind A/B listening study, arena Elo with 95% CI. Every confidence interval shown overlaps the leader, so these four are a statistical tie and none is the German winner. Hume octave-2 also ties on score but is not on the Speko roster.",
          "caption": "statistical tie, no winner"
        }
      ]
    },
    {
      "code": "fr",
      "label": "French",
      "legs": [
        {
          "mod": "STT",
          "picks": [
            {
              "provider": "ElevenLabs",
              "model": "scribe_v2_realtime",
              "id": "elevenlabs:scribe_v2_realtime",
              "value": 7.8
            },
            {
              "provider": "Soniox",
              "model": "stt-rt-v5",
              "id": "soniox:stt-rt-v5",
              "value": 8.8
            }
          ],
          "unit": "% WER",
          "note": "English-French code-switch. ElevenLabs 7.8% and Soniox 8.8%, both at 89% both-language coverage.",
          "caption": "EN-FR code-switch, no CI published"
        },
        {
          "mod": "LLM",
          "picks": [
            {
              "provider": "Anthropic",
              "model": "Claude Haiku 4.5",
              "id": "anthropic:claude-haiku-4-5",
              "value": 82.46
            }
          ],
          "unit": "%",
          "note": "The same recommendation on every language, carried across from our LLM board. Task completion is now measured per language: 21 models across ten languages, 19 scenarios at 3 iterations each, with a confidence interval on every cell — see the Multilingual LLM board. Nothing scores 100%; the English leader is gpt-5.6-luna at 94.21%, and the ten-language means run 68.8% to 74.0%, so language moves this axis far less than model choice does. Claude Haiku 4.5 is picked for behaviour rather than for topping the axis: the board publishes no composite score and no board-wide rank.",
          "caption": "measured in English"
        },
        {
          "mod": "TTS",
          "picks": [
            {
              "provider": "Cartesia",
              "model": "sonic-3.5",
              "id": "cartesia:sonic-3.5",
              "value": 1622
            },
            {
              "provider": "ElevenLabs",
              "model": "eleven_v3_conversational",
              "id": "elevenlabs:eleven_v3_conversational",
              "value": 1775
            },
            {
              "provider": "Soniox",
              "model": "tts-rt-v1",
              "id": "soniox:tts-rt-v1",
              "value": 1628
            },
            {
              "provider": "xAI Grok",
              "model": "grok-tts",
              "id": "xai:tts",
              "value": 1692
            }
          ],
          "unit": "Elo",
          "note": "Blind A/B listening study, arena Elo with 95% CI. The widest tie band on the board: the leader interval spans 232 Elo, so every system shown overlaps it and none of them is the French winner. Pick on latency and cost, not on this number. Hume octave-2 and Inworld inworld-tts-2 also tie on score; Hume is not on the Speko roster and Inworld is English-only through the gateway.",
          "caption": "statistical tie, no winner"
        }
      ]
    },
    {
      "code": "ar",
      "label": "Arabic",
      "legs": [
        {
          "mod": "STT",
          "picks": [],
          "unit": "% WER",
          "note": "No Arabic measurement exists on the streaming path. The Arabic board at /stt-multilingual is batch, is scored by character error rate rather than word error rate, and its leader has no valid streaming number on our gateway.",
          "blank": "no streaming measurement"
        },
        {
          "mod": "LLM",
          "picks": [
            {
              "provider": "Anthropic",
              "model": "Claude Haiku 4.5",
              "id": "anthropic:claude-haiku-4-5",
              "value": 82.46
            }
          ],
          "unit": "%",
          "note": "The same recommendation on every language, carried across from our LLM board. Task completion is now measured per language: 21 models across ten languages, 19 scenarios at 3 iterations each, with a confidence interval on every cell — see the Multilingual LLM board. Nothing scores 100%; the English leader is gpt-5.6-luna at 94.21%, and the ten-language means run 68.8% to 74.0%, so language moves this axis far less than model choice does. Claude Haiku 4.5 is picked for behaviour rather than for topping the axis: the board publishes no composite score and no board-wide rank.",
          "caption": "measured in English"
        },
        {
          "mod": "TTS",
          "picks": [
            {
              "provider": "Cartesia",
              "model": "sonic-3.5",
              "id": "cartesia:sonic-3.5",
              "value": 1644
            },
            {
              "provider": "xAI Grok",
              "model": "grok-tts",
              "id": "xai:tts",
              "value": 1736
            }
          ],
          "unit": "Elo",
          "note": "Blind A/B listening study, arena Elo with 95% CI; the smallest rater panel on the board. Only four systems ran, so the blanks elsewhere are not losses. One FLEURS Arabic corpus, no dialect split. ElevenLabs, which leads Spanish, German and French, falls below the leader here at 1564.",
          "caption": "statistical tie, no winner"
        }
      ]
    },
    {
      "code": "fil",
      "label": "Filipino",
      "legs": [
        {
          "mod": "STT",
          "picks": [],
          "unit": "% WER",
          "note": "Native Filipino streaming eval, n=30: ElevenLabs Scribe v2 Realtime measured 11.7% WER (95% CI 8.9-14.7) on 2026-07-19. It is the only measured Filipino STT row on this site, so there is no comparative ranking.",
          "blank": "not measured"
        },
        {
          "mod": "LLM",
          "picks": [
            {
              "provider": "Anthropic",
              "model": "Claude Haiku 4.5",
              "id": "anthropic:claude-haiku-4-5",
              "value": 82.46
            }
          ],
          "unit": "%",
          "note": "The same recommendation on every language, carried across from our LLM board. Task completion is now measured per language: 21 models across ten languages, 19 scenarios at 3 iterations each, with a confidence interval on every cell — see the Multilingual LLM board. Nothing scores 100%; the English leader is gpt-5.6-luna at 94.21%, and the ten-language means run 68.8% to 74.0%, so language moves this axis far less than model choice does. Claude Haiku 4.5 is picked for behaviour rather than for topping the axis: the board publishes no composite score and no board-wide rank.",
          "caption": "measured in English"
        },
        {
          "mod": "TTS",
          "picks": [
            {
              "provider": "Cartesia",
              "model": "sonic-3.5",
              "id": "cartesia:sonic-3.5",
              "value": 1597
            },
            {
              "provider": "ElevenLabs",
              "model": "eleven_v3_conversational",
              "id": "elevenlabs:eleven_v3_conversational",
              "value": 1693
            },
            {
              "provider": "OpenAI",
              "model": "gpt-4o-mini-tts",
              "id": "openai:gpt-4o-mini-tts",
              "value": 1676
            },
            {
              "provider": "xAI Grok",
              "model": "grok-tts",
              "id": "xai:tts",
              "value": 1657
            }
          ],
          "unit": "Elo",
          "note": "Blind A/B listening study, arena Elo with 95% CI; the thinnest study on the board. Every system shown cleared the measured Filipino language-identification gate and their intervals overlap, so none of them is the Filipino winner. MiniMax speech-2.8-hd, 1598 (1497-1717), also overlaps them and is not shown because the roster maps Alibaba to a different model. Gemini scored 2012 (1854-2289), above every system shown rather than tied with them; it was removed from the Speko roster on 2026-07-17 because its adapter returns one full clip instead of streaming, and it never ran the gate.",
          "caption": "statistical tie, no winner"
        }
      ]
    },
    {
      "code": "nb",
      "label": "Norwegian",
      "legs": [
        {
          "mod": "STT",
          "picks": [
            {
              "provider": "ElevenLabs",
              "model": "Scribe v2 Realtime",
              "id": "elevenlabs:scribe_v2_realtime",
              "value": 8.3
            },
            {
              "provider": "Soniox",
              "model": "stt-rt-v5",
              "id": "soniox:stt-rt-v5",
              "value": 8
            }
          ],
          "unit": "% WER",
          "note": "Monolingual FLEURS nb_no, n=50 clips, pooled corpus WER, measured over the live gateway WebSocket. Soniox 8.0% (95% CI 6.2-10.0) and ElevenLabs Scribe v2 Realtime 8.3% (6.3-10.5) overlap, so neither leads. OpenAI is absent because our own streaming adapter commits audio too early, not because the model is weak. Norwegian is the one language whose board also carries a batch arm on the same 50 clips (leader 5.6%, ElevenLabs Scribe v2, 2026-08-21) - it is the lower number but not the one a call runs on, and the streaming socket serves scribe_v2_realtime for any ElevenLabs pin regardless. The pick above stays on the streaming measurement.",
          "caption": "statistical tie, no winner"
        },
        {
          "mod": "LLM",
          "picks": [
            {
              "provider": "Anthropic",
              "model": "Claude Haiku 4.5",
              "id": "anthropic:claude-haiku-4-5",
              "value": 82.46
            }
          ],
          "unit": "%",
          "note": "The same recommendation on every language, carried across from our LLM board. Task completion is now measured per language: 21 models across ten languages, 19 scenarios at 3 iterations each, with a confidence interval on every cell — see the Multilingual LLM board. Nothing scores 100%; the English leader is gpt-5.6-luna at 94.21%, and the ten-language means run 68.8% to 74.0%, so language moves this axis far less than model choice does. Claude Haiku 4.5 is picked for behaviour rather than for topping the axis: the board publishes no composite score and no board-wide rank.",
          "caption": "measured in English"
        },
        {
          "mod": "TTS",
          "picks": [
            {
              "provider": "Cartesia",
              "model": "sonic-3.5",
              "id": "cartesia:sonic-3.5",
              "value": 1836
            }
          ],
          "unit": "Elo",
          "note": "Blind A/B listening study over six systems, arena Elo with 95% CI. Cartesia clears ElevenLabs 1628 (95% CI 1533-1733). Gemini scored 1705 within a hair of it but does not stream on our gateway. Cartesia sonic-3.5 default voice drifts roughly 5x the rest of the field in every language we have probed for drift; Norwegian was not one of them."
        }
      ]
    },
    {
      "code": "zh",
      "label": "Mandarin",
      "legs": [
        {
          "mod": "STT",
          "picks": [
            {
              "provider": "Soniox",
              "model": "stt-rt-v5",
              "id": "soniox:stt-rt-v5",
              "value": 11.8
            }
          ],
          "unit": "% MTER",
          "note": "Mandarin-English code-switch. Soniox is the only provider that survives the pair: 11.8% mixed-token error at 88% both-language coverage. ElevenLabs anglicizes Mandarin, dropping it to English at 34% coverage, so it is not shown.",
          "caption": "EN-ZH code-switch, mixed-token error"
        },
        {
          "mod": "LLM",
          "picks": [
            {
              "provider": "Anthropic",
              "model": "Claude Haiku 4.5",
              "id": "anthropic:claude-haiku-4-5",
              "value": 82.46
            }
          ],
          "unit": "%",
          "note": "The same recommendation on every language, carried across from our LLM board. Task completion is now measured per language: 21 models across ten languages, 19 scenarios at 3 iterations each, with a confidence interval on every cell — see the Multilingual LLM board. Nothing scores 100%; the English leader is gpt-5.6-luna at 94.21%, and the ten-language means run 68.8% to 74.0%, so language moves this axis far less than model choice does. Claude Haiku 4.5 is picked for behaviour rather than for topping the axis: the board publishes no composite score and no board-wide rank.",
          "caption": "measured in English"
        },
        {
          "mod": "TTS",
          "picks": [
            {
              "provider": "Cartesia",
              "model": "sonic-3.5",
              "id": "cartesia:sonic-3.5",
              "value": 1564
            },
            {
              "provider": "ElevenLabs",
              "model": "eleven_v3_conversational",
              "id": "elevenlabs:eleven_v3_conversational",
              "value": 1675
            },
            {
              "provider": "Inworld",
              "model": "inworld-tts-2",
              "id": "inworld:inworld-tts-2",
              "value": 1688
            }
          ],
          "unit": "Elo",
          "note": "Blind A/B listening study (2026-08-24), arena Elo with 95% CI, raters recruited on first language = Mandarin specifically rather than the generic Chinese pool. The top three intervals overlap — Inworld 1688 (1629–1754), ElevenLabs 1675 (1587–1767) and Gemini 1649 (1586–1724) are a statistical tie and none is the Mandarin winner. Gemini is not shown: it was removed from the Speko roster on 2026-07-17 because its adapter returns one full clip instead of streaming. MiniMax, the home-market vendor, measured 1467 — below the tie band.",
          "blank": "no listening study",
          "caption": "statistical tie, no winner"
        }
      ]
    }
  ],
  "projector": [
    {
      "name": "deepgram:nova-3 · cerebras:gpt-oss-120b · elevenlabs:eleven_flash_v2_5",
      "c": 0.055,
      "fab": true
    },
    {
      "name": "deepgram:nova-3 · cerebras:gpt-oss-120b · cartesia:sonic-3.5",
      "c": 0.058
    },
    {
      "name": "deepgram:nova-3 · openai:gpt-4.1-mini · cartesia:sonic-3.5 (never books)",
      "c": 0.079,
      "fab": true
    },
    {
      "name": "cartesia:ink-2 · cerebras:gpt-oss-120b · cartesia:sonic-3.5",
      "c": 0.087
    },
    {
      "name": "deepgram:nova-3 · cerebras:gpt-oss-120b · minimax:speech-2.6-hd",
      "c": 0.092
    },
    {
      "name": "deepgram:nova-3 · openai:gpt-4.1 · cartesia:sonic-3.5",
      "c": 0.111
    },
    {
      "name": "deepgram:nova-3 · cerebras:gpt-oss-120b · deepgram:aura-2",
      "c": 0.123
    },
    {
      "name": "deepgram:nova-3 · openai:gpt-5-nano · cartesia:sonic-3.5",
      "c": 0.145
    },
    {
      "name": "elevenlabs:scribe_v2_realtime · openai:gpt-4.1 · cartesia:sonic-3.5",
      "c": 0.147
    },
    {
      "name": "elevenlabs:scribe_v2_realtime · openai:gpt-4.1 · elevenlabs:eleven_flash_v2_5",
      "c": 0.469
    }
  ],
  "turntakingSynthetic": [
    {
      "label": "Turnsense",
      "kind": "text",
      "acc": 98.8,
      "cut": 0
    },
    {
      "label": "LiveKit Intl",
      "kind": "text",
      "acc": 90.6,
      "cut": 6.2
    },
    {
      "label": "LiveKit EN",
      "kind": "text",
      "acc": 79.2,
      "cut": 26.2
    },
    {
      "label": "Smart Turn",
      "kind": "audio",
      "acc": 62.4,
      "cut": 57.7
    }
  ],
  "turntakingProsody": [
    {
      "cue": "Final F0 slope",
      "unit": "Hz/s",
      "end": -119.7,
      "wait": 42.9,
      "want": "complete falls, incomplete rises/holds"
    },
    {
      "cue": "Final energy slope",
      "unit": "dB/s",
      "end": -14.9,
      "wait": -2.5,
      "want": "complete tapers off"
    }
  ]
}