{"observation":{"id":"311e6d75-7711-42c7-9249-985257561481","tool":"aws-transcribe","tool_name":"AWS Transcribe","criterion":"input-handling","criterion_name":"Input handling","criterion_definition":"Whether the tool accepts the benchmark audio as provided and processes it end to end without objection, including the file size/duration it can handle.","criterion_evidence_type":"capability","criterion_rank_role":"context","criterion_rank_role_reason":"Being able to accept the benchmark audio and handle its size/duration is necessary to test the tool, but it is an operational constraint rather than the transcription quality being ranked. (3 of 3 judges)","scenario":"overlapping-meeting-speech-with-cross-talk","scenario_name":"Overlapping meeting speech with cross-talk","group_tag":"speech-to-text-benchmark","scenario_description":"A long AMI meeting audio file with multiple speakers talking over one another, background room noise, and crosstalk. It was used to test how well an STT system handles noisy multi-speaker conversational audio and speaker separation.","modality":"audio","input_text":null,"input_artifact_refs":[{"alt":null,"url":"https://cdn.futuresmart.ai/public/aidemos/12990982a1614cc5b18a3a450e0e657f.wav?v=1","role":"input","filename":"crosstalk.wav"}],"stresses":["overlapping speech","background noise robustness","multi-speaker separation","speaker diarization accuracy","long-form audio handling"],"verdict":"worked","score":null,"score_total":null,"note":"Accepts a 65.39 MB, 2142.709 s WAV and processes it through without objection.","evidence_state":"verified","source":null,"artifacts":[{"url":"https://cdn.futuresmart.ai/public/aidemos/3d810cf7afca4101b2648301b88e0dce.png?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/47be998587d84e5ba0c33ef977d1569d.png?v=1","role":"context","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/439691e323bc456a84344fab78c75c30.png?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/a8952ff0bc8a4c7ebadaf9ec2df44ca9.png?v=1","role":"context","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/892fcc8e770c4aa5a0802bc2cb1eaeee.png?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/652e08c8c4b64c588cd4f3a90c91bf9f.png?v=1","role":"context","alt":null}],"run_id":"469de0c2-d727-4f8f-a60e-e3a5bf8e8588","study_title":"Transcribe Audio Accurately — Speech-to-Text Engine Benchmark","study_kind":"generation","research_task":"86baxegpu","tested_at":null,"completeness":"input-and-output","input":{"state":"files","text":null,"files":[{"url":"https://cdn.futuresmart.ai/public/aidemos/12990982a1614cc5b18a3a450e0e657f.wav?v=1","filename":"crosstalk.wav","alt":"Overlapping meeting speech with cross-talk","role":"input"}],"modality":"audio","stresses":["overlapping speech","background noise robustness","multi-speaker separation","speaker diarization accuracy","long-form audio handling"]},"tool_page_slug":"aws-transcribe","tool_url":"https://aidemos.com/tools/aws-transcribe","permalink":"https://aidemos.com/evidence/311e6d75-7711-42c7-9249-985257561481","api_url":"https://ai.aidemos.com/v1/observations/311e6d75-7711-42c7-9249-985257561481"},"peers":[{"id":"fb95bb1d-6e61-4356-ad55-4b2e77730f17","tool":"elevenlabs-scribe","tool_name":"ElevenLabs Scribe","verdict":"worked","score":null,"score_total":null,"note":"The tool accepts a 65.39 MB, 2142.709 s audio file and processes it end to end without objection.","artifact_count":6,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/666aae8148fd4eea870ff63c3327d3cb.png?v=1","evidence_url":"https://aidemos.com/evidence/fb95bb1d-6e61-4356-ad55-4b2e77730f17"}],"other_criteria":[{"id":"4962545f-f4ad-47ca-8668-c44f159ada99","criterion":"export","criterion_name":"Export","rank_role":"context","verdict":"worked","score":3.0,"score_total":3.0,"note":"Returns a full developer payload with payload depth 3/3, 5912 word-level timed tokens, confidence values, and speaker labels; the raw JSON walk spans 7 levels and 19457 objects.","artifact_count":3,"evidence_url":"https://aidemos.com/evidence/4962545f-f4ad-47ca-8668-c44f159ada99"},{"id":"8b0f2b7f-f263-4002-88ec-ccda7fa02a4e","criterion":"output-quality","criterion_name":"Output quality","rank_role":"decisive","verdict":"struggled","score":33.88,"score_total":null,"note":"Transcript quality is weak at 33.88% WER, with 363 substitutions, 2162 deletions, and 43 insertions against 7579 reference words.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/8b0f2b7f-f263-4002-88ec-ccda7fa02a4e"}],"appears_in":[{"page_type":"ranking","slug":"speech-to-text-apis","title":"Best AI Tools for Accurate Speech-to-Text on Hard Audio","url":"https://aidemos.com/best/speech-to-text-apis","binding":"run"}],"same_scenario":[{"id":"fb95bb1d-6e61-4356-ad55-4b2e77730f17","tool":"elevenlabs-scribe","tool_name":"ElevenLabs Scribe","verdict":"worked","score":null,"score_total":null,"note":"The tool accepts a 65.39 MB, 2142.709 s audio file and processes it end to end without objection."}]}