{"observation":{"id":"fb95bb1d-6e61-4356-ad55-4b2e77730f17","tool":"elevenlabs-scribe","tool_name":"ElevenLabs Scribe","criterion":"input-handling","criterion_name":"Input handling","criterion_definition":"Whether the tool accepts the benchmark audio as provided and processes it end to end without objection, including the file size/duration it can handle.","criterion_evidence_type":"capability","criterion_rank_role":"context","criterion_rank_role_reason":"Being able to accept the benchmark audio and handle its size/duration is necessary to test the tool, but it is an operational constraint rather than the transcription quality being ranked. (3 of 3 judges)","scenario":"overlapping-meeting-speech-with-cross-talk","scenario_name":"Overlapping meeting speech with cross-talk","group_tag":"speech-to-text-benchmark","scenario_description":"A long AMI meeting audio file with multiple speakers talking over one another, background room noise, and crosstalk. It was used to test how well an STT system handles noisy multi-speaker conversational audio and speaker separation.","modality":"audio","input_text":null,"input_artifact_refs":[{"alt":null,"url":"https://cdn.futuresmart.ai/public/aidemos/12990982a1614cc5b18a3a450e0e657f.wav?v=1","role":"input","filename":"crosstalk.wav"}],"stresses":["overlapping speech","background noise robustness","multi-speaker separation","speaker diarization accuracy","long-form audio handling"],"verdict":"worked","score":null,"score_total":null,"note":"The tool accepts a 65.39 MB, 2142.709 s audio file and processes it end to end without objection.","evidence_state":"verified","source":null,"artifacts":[{"url":"https://cdn.futuresmart.ai/public/aidemos/666aae8148fd4eea870ff63c3327d3cb.png?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/afa9e0579a6b4d3687bd1e7729b54546.png?v=1","role":"context","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/3fc6bef146c8454bb016e0f12672c04c.png?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/799fe3740fed4fad962f4d443f4e44d2.png?v=1","role":"context","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/0f370484b7634847b7a0b078a7a3082e.png?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/1cdb5610e7af4f10abf1e8995c3178f0.png?v=1","role":"context","alt":null}],"run_id":"469de0c2-d727-4f8f-a60e-e3a5bf8e8588","study_title":"Transcribe Audio Accurately — Speech-to-Text Engine Benchmark","study_kind":"generation","research_task":"86baxegpu","tested_at":null,"completeness":"input-and-output","input":{"state":"files","text":null,"files":[{"url":"https://cdn.futuresmart.ai/public/aidemos/12990982a1614cc5b18a3a450e0e657f.wav?v=1","filename":"crosstalk.wav","alt":"Overlapping meeting speech with cross-talk","role":"input"}],"modality":"audio","stresses":["overlapping speech","background noise robustness","multi-speaker separation","speaker diarization accuracy","long-form audio handling"]},"tool_page_slug":"elevenlabs-scribe","tool_url":"https://aidemos.com/tools/elevenlabs-scribe","permalink":"https://aidemos.com/evidence/fb95bb1d-6e61-4356-ad55-4b2e77730f17","api_url":"https://ai.aidemos.com/v1/observations/fb95bb1d-6e61-4356-ad55-4b2e77730f17"},"peers":[{"id":"311e6d75-7711-42c7-9249-985257561481","tool":"aws-transcribe","tool_name":"AWS Transcribe","verdict":"worked","score":null,"score_total":null,"note":"Accepts a 65.39 MB, 2142.709 s WAV and processes it through without objection.","artifact_count":6,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/3d810cf7afca4101b2648301b88e0dce.png?v=1","evidence_url":"https://aidemos.com/evidence/311e6d75-7711-42c7-9249-985257561481"}],"other_criteria":[{"id":"a9c3d354-c6ec-4468-ba77-6df4de2e7c41","criterion":"export","criterion_name":"Export","rank_role":"context","verdict":"worked","score":3.0,"score_total":3.0,"note":"The returned payload is rich and fully structured at 3/3 depth, with word-level timestamps, confidence values, speaker labels, and 14,506 timed tokens.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/a9c3d354-c6ec-4468-ba77-6df4de2e7c41"},{"id":"667df61a-e337-48ae-b65c-6d910e2ccf83","criterion":"output-quality","criterion_name":"Output quality","rank_role":"decisive","verdict":"struggled","score":26.67,"score_total":null,"note":"Transcript accuracy is weak on crosstalk, with WER 26.67% and 856 substitutions, 782 deletions, and 383 insertions over a 7,579-word reference.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/667df61a-e337-48ae-b65c-6d910e2ccf83"}],"appears_in":[{"page_type":"ranking","slug":"speech-to-text-apis","title":"Best AI Tools for Accurate Speech-to-Text on Hard Audio","url":"https://aidemos.com/best/speech-to-text-apis","binding":"run"}],"same_scenario":[{"id":"311e6d75-7711-42c7-9249-985257561481","tool":"aws-transcribe","tool_name":"AWS Transcribe","verdict":"worked","score":null,"score_total":null,"note":"Accepts a 65.39 MB, 2142.709 s WAV and processes it through without objection."}]}