{"observation":{"id":"9aaccf1c-90fc-4c62-acb9-43e41a10110c","tool":"deepgram","tool_name":"Deepgram","criterion":"export","criterion_name":"Export","criterion_definition":"How complete the returned transcript payload is, as reflected in the depth or richness of what the tool outputs.","criterion_evidence_type":"transformation","criterion_rank_role":"context","criterion_rank_role_reason":"Transcript payload richness is useful for comparison, but it does not measure transcription correctness against the reference. (3 of 3 judges)","scenario":"medical-anatomy-narration-with-dense-jargon","scenario_name":"Medical anatomy narration with dense jargon","group_tag":"speech-to-text-benchmark","scenario_description":"A long narrated excerpt from Henry Gray's Anatomy of the Human Body containing dense medical terminology and accented articulation. It was used to test lexical accuracy on domain-specific jargon and spelling of technical terms.","modality":"audio","input_text":null,"input_artifact_refs":[{"alt":null,"url":"https://cdn.futuresmart.ai/public/aidemos/071409ab3a0d4358bb238133eafcb60b.mp3?v=1","role":"input","filename":"medical_terms.mp3"}],"stresses":["domain-specific vocabulary recognition","medical term spelling accuracy","accented speech robustness","phoneme-to-grapheme precision","long-form audio handling"],"verdict":"worked","score":null,"score_total":null,"note":"Returns a rich developer payload on medical narration, with word-level timing, confidence, speaker labels, 5,845 timed tokens, 1 distinct speaker, payload depth 3/3, and JSON depth 11 across 5,881 objects.","evidence_state":"verified","source":null,"artifacts":[{"url":"https://cdn.futuresmart.ai/public/aidemos/53a62dd8311a4f0c9ff89e699bc07877.mp3?v=1","role":"input","alt":"53a62dd8311a4f0c9ff89e699bc07877.mp3"},{"url":"https://cdn.futuresmart.ai/public/aidemos/0a1e5543317c450eb410eaf28c70a62e.png?v=1","role":"output","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/553abf0d1f784b09bd25419d0ea2cc91.png?v=1","role":"context","alt":null},{"url":"https://d3epheqghktydj.cloudfront.net/research-media-raw-response-2-13925b55bd6c.json","role":"output","alt":null}],"run_id":"469de0c2-d727-4f8f-a60e-e3a5bf8e8588","study_title":"Transcribe Audio Accurately — Speech-to-Text Engine Benchmark","study_kind":"generation","research_task":"86baxegpu","tested_at":null,"completeness":"input-and-output","input":{"state":"files","text":null,"files":[{"url":"https://cdn.futuresmart.ai/public/aidemos/071409ab3a0d4358bb238133eafcb60b.mp3?v=1","filename":"medical_terms.mp3","alt":"Medical anatomy narration with dense jargon","role":"input"}],"modality":"audio","stresses":["domain-specific vocabulary recognition","medical term spelling accuracy","accented speech robustness","phoneme-to-grapheme precision","long-form audio handling"]},"tool_page_slug":"deepgram","tool_url":"https://aidemos.com/tools/deepgram","permalink":"https://aidemos.com/evidence/9aaccf1c-90fc-4c62-acb9-43e41a10110c","api_url":"https://ai.aidemos.com/v1/observations/9aaccf1c-90fc-4c62-acb9-43e41a10110c"},"peers":[{"id":"9c7fb0a9-ef0e-4b34-b866-e8084b7cf4fe","tool":"assemblyai","tool_name":"AssemblyAI","verdict":"worked","score":null,"score_total":null,"note":"Returns a rich developer payload with word-level timestamps, confidence values, and speaker labels; the raw response shows 5,437 timed tokens and JSON depth 5 across 5,443 objects, with payload depth 3/3.","artifact_count":3,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/4d74aea939f046b18bde5cd3b0764ce8.png?v=1","evidence_url":"https://aidemos.com/evidence/9c7fb0a9-ef0e-4b34-b866-e8084b7cf4fe"},{"id":"e0b1dbec-8c14-45ad-8004-3982946d8486","tool":"aws-transcribe","tool_name":"AWS Transcribe","verdict":"worked","score":3.0,"score_total":3.0,"note":"Returns a full developer payload with payload depth 3/3, 2829 word-level timed tokens, confidence values, and speaker labels; the raw JSON walk spans 7 levels and 9023 objects.","artifact_count":3,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/53a62dd8311a4f0c9ff89e699bc07877.mp3?v=1","evidence_url":"https://aidemos.com/evidence/e0b1dbec-8c14-45ad-8004-3982946d8486"},{"id":"47fc1f40-4845-448e-b664-bda89512feec","tool":"elevenlabs-scribe","tool_name":"ElevenLabs Scribe","verdict":"worked","score":3.0,"score_total":3.0,"note":"The returned payload is rich and fully structured at 3/3 depth, with word-level timestamps, confidence values, speaker labels, and 5,448 timed tokens.","artifact_count":2,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/315503dff77c49fa9f45a285edaf17fa.png?v=1","evidence_url":"https://aidemos.com/evidence/47fc1f40-4845-448e-b664-bda89512feec"},{"id":"6da4b91b-f0ff-4b4e-8813-ef33e15fe3e9","tool":"gladia","tool_name":"Gladia","verdict":"worked","score":null,"score_total":null,"note":"Returns a rich transcript payload rather than plain text: payload depth is 3/3, with word-level timing, confidence, speaker labels, 5,854 timed tokens, and 7-level JSON nesting across 5,862 objects.","artifact_count":4,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/53a62dd8311a4f0c9ff89e699bc07877.mp3?v=1","evidence_url":"https://aidemos.com/evidence/6da4b91b-f0ff-4b4e-8813-ef33e15fe3e9"},{"id":"edb2573e-59c0-4979-a4fe-8f0f627f3acd","tool":"google-cloud-speech-to-text","tool_name":"Google Cloud Speech-to-Text","verdict":"mixed","score":null,"score_total":null,"note":"Returns a mid-depth transcript payload: payload depth 2/3 with 2750 word-level timed tokens, confidence present, and no speaker labels.","artifact_count":2,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/b6563c28cf784b65bafd2dc663fbcabd.png?v=1","evidence_url":"https://aidemos.com/evidence/edb2573e-59c0-4979-a4fe-8f0f627f3acd"},{"id":"38781cd9-9d9c-4e6e-95c3-be85e653f18f","tool":"groqcloud","tool_name":"GroqCloud","verdict":"worked","score":null,"score_total":null,"note":"Returns a verbose JSON transcript payload with task, language, duration, segments, word timestamps, confidence data, and no speaker labels; the payload depth is 2/3 with 206 timed tokens.","artifact_count":3,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/a466cf3e61c947b48d0be7aeef96433e.png?v=1","evidence_url":"https://aidemos.com/evidence/38781cd9-9d9c-4e6e-95c3-be85e653f18f"},{"id":"5e4b9ba8-aac3-4559-84fe-74ea4a00a3fa","tool":"openai-speech-to-text","tool_name":"OpenAI Speech-to-Text","verdict":"mixed","score":null,"score_total":null,"note":"Returns a fairly rich verbose_json payload with 2664 word-level timed tokens, but no confidence values and no speaker labels; JSON depth is 3 with 2666 objects.","artifact_count":1,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/13a31fd6f7fe4d9fa7942ad128d39536.png?v=1","evidence_url":"https://aidemos.com/evidence/5e4b9ba8-aac3-4559-84fe-74ea4a00a3fa"},{"id":"4269cac8-8737-4d14-9271-14dbac75308d","tool":"rev-ai","tool_name":"Rev AI","verdict":"worked","score":null,"score_total":null,"note":"Returned a full developer payload with word-level timing, confidence and speaker labels; the raw response walk reports 2779 timed tokens, payload depth 3/3, and JSON depth 5 across 5873 objects.","artifact_count":3,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/53a62dd8311a4f0c9ff89e699bc07877.mp3?v=1","evidence_url":"https://aidemos.com/evidence/4269cac8-8737-4d14-9271-14dbac75308d"},{"id":"3bfc6b47-2d58-42b0-865a-b92338de1919","tool":"speechmatics","tool_name":"Speechmatics","verdict":"worked","score":null,"score_total":null,"note":"Returns a rich transcript payload with payload_depth 3/3, including word-level timing, confidence values, and speaker labels in the JSON response.","artifact_count":4,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/845fb0bf5bbd4d9e94e2645cdd415692.png?v=1","evidence_url":"https://aidemos.com/evidence/3bfc6b47-2d58-42b0-865a-b92338de1919"}],"other_criteria":[{"id":"9e0d991b-d39f-4f0f-b7c0-ca643271064e","criterion":"output-quality","criterion_name":"Output quality","rank_role":"decisive","verdict":"worked","score":5.43,"score_total":null,"note":"Keeps lexical accuracy high on dense medical narration: WER is 5.43% with 82 substitutions, 34 deletions, and 32 insertions against a 2728-word reference, and jargon recall is 100.0% (9/9 scored terms).","artifact_count":3,"evidence_url":"https://aidemos.com/evidence/9e0d991b-d39f-4f0f-b7c0-ca643271064e"}],"appears_in":[{"page_type":"ranking","slug":"speech-to-text-apis","title":"Best AI Tools for Accurate Speech-to-Text on Hard Audio","url":"https://aidemos.com/best/speech-to-text-apis","binding":"run"}],"same_scenario":[{"id":"9c7fb0a9-ef0e-4b34-b866-e8084b7cf4fe","tool":"assemblyai","tool_name":"AssemblyAI","verdict":"worked","score":null,"score_total":null,"note":"Returns a rich developer payload with word-level timestamps, confidence values, and speaker labels; the raw response shows 5,437 timed tokens and JSON depth 5 across 5,443 objects, with payload depth 3/3."},{"id":"e0b1dbec-8c14-45ad-8004-3982946d8486","tool":"aws-transcribe","tool_name":"AWS Transcribe","verdict":"worked","score":3.0,"score_total":3.0,"note":"Returns a full developer payload with payload depth 3/3, 2829 word-level timed tokens, confidence values, and speaker labels; the raw JSON walk spans 7 levels and 9023 objects."},{"id":"47fc1f40-4845-448e-b664-bda89512feec","tool":"elevenlabs-scribe","tool_name":"ElevenLabs Scribe","verdict":"worked","score":3.0,"score_total":3.0,"note":"The returned payload is rich and fully structured at 3/3 depth, with word-level timestamps, confidence values, speaker labels, and 5,448 timed tokens."},{"id":"6da4b91b-f0ff-4b4e-8813-ef33e15fe3e9","tool":"gladia","tool_name":"Gladia","verdict":"worked","score":null,"score_total":null,"note":"Returns a rich transcript payload rather than plain text: payload depth is 3/3, with word-level timing, confidence, speaker labels, 5,854 timed tokens, and 7-level JSON nesting across 5,862 objects."},{"id":"edb2573e-59c0-4979-a4fe-8f0f627f3acd","tool":"google-cloud-speech-to-text","tool_name":"Google Cloud Speech-to-Text","verdict":"mixed","score":null,"score_total":null,"note":"Returns a mid-depth transcript payload: payload depth 2/3 with 2750 word-level timed tokens, confidence present, and no speaker labels."},{"id":"38781cd9-9d9c-4e6e-95c3-be85e653f18f","tool":"groqcloud","tool_name":"GroqCloud","verdict":"worked","score":null,"score_total":null,"note":"Returns a verbose JSON transcript payload with task, language, duration, segments, word timestamps, confidence data, and no speaker labels; the payload depth is 2/3 with 206 timed tokens."},{"id":"5e4b9ba8-aac3-4559-84fe-74ea4a00a3fa","tool":"openai-speech-to-text","tool_name":"OpenAI Speech-to-Text","verdict":"mixed","score":null,"score_total":null,"note":"Returns a fairly rich verbose_json payload with 2664 word-level timed tokens, but no confidence values and no speaker labels; JSON depth is 3 with 2666 objects."},{"id":"4269cac8-8737-4d14-9271-14dbac75308d","tool":"rev-ai","tool_name":"Rev AI","verdict":"worked","score":null,"score_total":null,"note":"Returned a full developer payload with word-level timing, confidence and speaker labels; the raw response walk reports 2779 timed tokens, payload depth 3/3, and JSON depth 5 across 5873 objects."},{"id":"3bfc6b47-2d58-42b0-865a-b92338de1919","tool":"speechmatics","tool_name":"Speechmatics","verdict":"worked","score":null,"score_total":null,"note":"Returns a rich transcript payload with payload_depth 3/3, including word-level timing, confidence values, and speaker labels in the JSON response."}]}