{"observation":{"id":"7ca3e91d-1ecd-4090-b774-95b3fdcace36","tool":"groqcloud","tool_name":"GroqCloud","criterion":"automation-level","criterion_name":"Automation level","criterion_definition":"How many API steps or calls the workflow requires, and whether it completes without operator input.","criterion_evidence_type":"capability","criterion_rank_role":"context","criterion_rank_role_reason":"How many API steps or whether operator input is needed affects convenience and workflow, but not whether the engine transcribes accurately once run. (3 of 3 judges)","scenario":"medical-anatomy-narration-with-dense-jargon","scenario_name":"Medical anatomy narration with dense jargon","group_tag":"speech-to-text-benchmark","scenario_description":"A long narrated excerpt from Henry Gray's Anatomy of the Human Body containing dense medical terminology and accented articulation. It was used to test lexical accuracy on domain-specific jargon and spelling of technical terms.","modality":"audio","input_text":null,"input_artifact_refs":[{"alt":null,"url":"https://cdn.futuresmart.ai/public/aidemos/071409ab3a0d4358bb238133eafcb60b.mp3?v=1","role":"input","filename":"medical_terms.mp3"}],"stresses":["domain-specific vocabulary recognition","medical term spelling accuracy","accented speech robustness","phoneme-to-grapheme precision","long-form audio handling"],"verdict":"worked","score":null,"score_total":null,"note":"Completes transcription as a single POST with the response returned inline and no operator intervention; the run trace notes that per-call timings were not instrumented.","evidence_state":"verified","source":null,"artifacts":[{"url":"https://cdn.futuresmart.ai/public/aidemos/f03fbeee36ef4a56931effa9f70be63f.png?v=1","role":"output","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/a466cf3e61c947b48d0be7aeef96433e.png?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/b83ec0926d884173bc6a385931d0cc39.png?v=1","role":"output","alt":null}],"run_id":"469de0c2-d727-4f8f-a60e-e3a5bf8e8588","study_title":"Transcribe Audio Accurately — Speech-to-Text Engine Benchmark","study_kind":"generation","research_task":"86baxegpu","tested_at":null,"completeness":"input-and-output","input":{"state":"files","text":null,"files":[{"url":"https://cdn.futuresmart.ai/public/aidemos/071409ab3a0d4358bb238133eafcb60b.mp3?v=1","filename":"medical_terms.mp3","alt":"Medical anatomy narration with dense jargon","role":"input"}],"modality":"audio","stresses":["domain-specific vocabulary recognition","medical term spelling accuracy","accented speech robustness","phoneme-to-grapheme precision","long-form audio handling"]},"tool_page_slug":"groqcloud","tool_url":"https://aidemos.com/tools/groqcloud","permalink":"https://aidemos.com/evidence/7ca3e91d-1ecd-4090-b774-95b3fdcace36","api_url":"https://ai.aidemos.com/v1/observations/7ca3e91d-1ecd-4090-b774-95b3fdcace36"},"peers":[{"id":"677302a8-3a2d-4042-9b1e-5bc8ae009635","tool":"google-cloud-speech-to-text","tool_name":"Google Cloud Speech-to-Text","verdict":"mixed","score":null,"score_total":null,"note":"The recorded workflow submits the audio as a POST to the v2 endpoint and reaches a scored result without operator input, but the trace explicitly says per-call timings were not instrumented, so no measured call count is claimed.","artifact_count":3,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/4d93c717959246b9836fd7714218b8b2.png?v=1","evidence_url":"https://aidemos.com/evidence/677302a8-3a2d-4042-9b1e-5bc8ae009635"}],"other_criteria":[{"id":"38781cd9-9d9c-4e6e-95c3-be85e653f18f","criterion":"export","criterion_name":"Export","rank_role":"context","verdict":"worked","score":null,"score_total":null,"note":"Returns a verbose JSON transcript payload with task, language, duration, segments, word timestamps, confidence data, and no speaker labels; the payload depth is 2/3 with 206 timed tokens.","artifact_count":3,"evidence_url":"https://aidemos.com/evidence/38781cd9-9d9c-4e6e-95c3-be85e653f18f"},{"id":"65a93d95-5cb0-41c2-96df-b2947129dba6","criterion":"output-quality","criterion_name":"Output quality","rank_role":"decisive","verdict":"worked","score":3.15,"score_total":null,"note":"Delivers high lexical accuracy on dense medical narration: WER 3.15% with 58 substitutions, 13 deletions, and 15 insertions over 2728 reference words, returning 2730 hypothesis words.","artifact_count":3,"evidence_url":"https://aidemos.com/evidence/65a93d95-5cb0-41c2-96df-b2947129dba6"}],"appears_in":[{"page_type":"ranking","slug":"speech-to-text-apis","title":"Best AI Tools for Accurate Speech-to-Text on Hard Audio","url":"https://aidemos.com/best/speech-to-text-apis","binding":"run"}],"same_scenario":[{"id":"677302a8-3a2d-4042-9b1e-5bc8ae009635","tool":"google-cloud-speech-to-text","tool_name":"Google Cloud Speech-to-Text","verdict":"mixed","score":null,"score_total":null,"note":"The recorded workflow submits the audio as a POST to the v2 endpoint and reaches a scored result without operator input, but the trace explicitly says per-call timings were not instrumented, so no measured call count is claimed."}]}