{"observation":{"id":"73354714-3c90-4a35-a5c6-c27f03292e80","tool":"reducto","tool_name":"Reducto","criterion":"advanced-features","criterion_name":"Advanced Features","criterion_definition":"Provides separate table/chart extraction and flags low-confidence OCR or ambiguous regions.","criterion_evidence_type":"transformation","criterion_rank_role":"context","criterion_rank_role_reason":"Separate extraction modes and OCR confidence flags are useful workflow features, but they do not by themselves determine whether the Markdown conversion is good. (3 of 3 judges)","scenario":"scanned-research-paper","scenario_name":"Scanned Research Paper","group_tag":"scanned-research-paper","scenario_description":"An image-only scanned research paper used to stress OCR and layout recovery in a multi-column academic document with figures, charts, tables, captions, and references.","modality":"pdf","input_text":null,"input_artifact_refs":[{"alt":null,"url":"https://d3epheqghktydj.cloudfront.net/convert-a-complex-pdf-into-clean-markdow-scanned-research-pdf-7b86de49784d.pdf","role":"input","filename":"Scanned Research PDF.pdf"}],"stresses":["OCR on scanned pages","Multi-column reading order","Figure and chart handling","Table reconstruction from scans","Caption association","Reference extraction","Overall document structure retention"],"verdict":"worked","score":null,"score_total":null,"note":"Schema-driven extract.run fully recovers the corrupted Table 4 rows, returning all 12 checked rows exactly, including the 10-inch-cut 1979 MPB row whose parse.run output was badly garbled.","evidence_state":"verified","source":null,"artifacts":[{"url":"https://cdn.futuresmart.ai/public/aidemos/7da94747d02a4142be818fe0b54106fb.png?v=1","role":"output","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/6f8f6fb4e5a14d24a4ccb8a8fa05f5a0.png?v=1","role":"output","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/060335669f5743f7bfb10af39347ad67.png?v=1","role":"input","alt":null}],"run_id":"6e3160de-fe46-4b45-b071-72560b5c5d0e","study_title":"Convert a Complex PDF into Clean Markdown with an API","study_kind":"generation","research_task":"86b9h7t37","tested_at":null,"completeness":"input-and-output","input":{"state":"files","text":null,"files":[{"url":"https://d3epheqghktydj.cloudfront.net/convert-a-complex-pdf-into-clean-markdow-scanned-research-pdf-7b86de49784d.pdf","filename":"Scanned Research PDF.pdf","alt":"Scanned Research Paper","role":"input"}],"modality":"pdf","stresses":["OCR on scanned pages","Multi-column reading order","Figure and chart handling","Table reconstruction from scans","Caption association","Reference extraction","Overall document structure retention"]},"tool_page_slug":"reducto","tool_url":"https://aidemos.com/tools/reducto","permalink":"https://aidemos.com/evidence/73354714-3c90-4a35-a5c6-c27f03292e80","api_url":"https://ai.aidemos.com/v1/observations/73354714-3c90-4a35-a5c6-c27f03292e80"},"peers":[{"id":"347aaab3-eada-4d58-a2c1-5c5c029de742","tool":"tensorlake","tool_name":"Tensorlake","verdict":"worked","score":null,"score_total":null,"note":"Performs dedicated chart extraction on a scanned bar chart, turning the figure into structured chart content with year-by-year values, treatment labels, and the 'CUT COMPLETED' annotation.","artifact_count":2,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/de1454c783be40d6ae9c9be1b56215f5.png?v=1","evidence_url":"https://aidemos.com/evidence/347aaab3-eada-4d58-a2c1-5c5c029de742"},{"id":"284ba196-edbd-492d-a9db-504bfe85f361","tool":"upstage-ai","tool_name":"Upstage AI","verdict":"worked","score":null,"score_total":null,"note":"Separately extracts Figure 3 into year-by-year values for 1972–1981 and tags it as an image asset, showing chart-specific extraction beyond plain OCR.","artifact_count":2,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/07b2a69c1a724a928dca5bb93ef9959b.png?v=1","evidence_url":"https://aidemos.com/evidence/284ba196-edbd-492d-a9db-504bfe85f361"}],"other_criteria":[{"id":"3bec7ed4-9b8e-44f3-b548-b76f0215b34c","criterion":"complex-document-handling","criterion_name":"Complex Document Handling","rank_role":"decisive","verdict":"struggled","score":null,"score_total":null,"note":"Handles the 12-page scan in 10.6 seconds with no truncation, but the densest 17-column table has unreadable stretches with injected glyphs and run-on numbers.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/3bec7ed4-9b8e-44f3-b548-b76f0215b34c"},{"id":"39ecbd6f-1599-4ba1-ae17-16b29b9c07cb","criterion":"markdown-quality","criterion_name":"Markdown Quality","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Uses no invented tags or malformed markdown; the only HTML seen is legitimate <br /> inside table cells, so the syntax stays clean even though the document does not surface real heading markup.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/39ecbd6f-1599-4ba1-ae17-16b29b9c07cb"},{"id":"7840c24e-489b-4551-b576-148911ebe640","criterion":"reading-order-structure","criterion_name":"Reading Order & Structure","rank_role":"decisive","verdict":"struggled","score":null,"score_total":null,"note":"Displaces the byline by a full column: the author line that sits above the two-column split in the source is emitted only after the entire left column, and its footnote markers are rendered inconsistently.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/7840c24e-489b-4551-b576-148911ebe640"},{"id":"671f1c02-54bd-415d-8592-f65b9923c375","criterion":"table-preservation","criterion_name":"Table Preservation","rank_role":"decisive","verdict":"failed","score":null,"score_total":null,"note":"Catastrophically corrupts the dense 17-column Table 4, injecting non-Latin glyphs into numeric cells and producing run-on strings such as 238-563NT, 2231, and 8998 6 in the 10-inch and 12-inch blocks.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/671f1c02-54bd-415d-8592-f65b9923c375"},{"id":"fcad1a88-bccf-4eea-8bf5-0d8e5dbc22ee","criterion":"table-preservation","criterion_name":"Table Preservation","rank_role":"decisive","verdict":"mixed","score":null,"score_total":null,"note":"Partially reconstructs Table 1: most of the roughly 90 numeric values are exact, but literal 0 values in the 12-inch column become blanks, one mean cell picks up stray digits (33.0 830000), and a row-label-only section header is broadcast across all six columns in one instance.","artifact_count":3,"evidence_url":"https://aidemos.com/evidence/fcad1a88-bccf-4eea-8bf5-0d8e5dbc22ee"},{"id":"b0876a41-66cd-455c-b38c-421968bf4e16","criterion":"text-ocr-completeness","criterion_name":"Text & OCR Completeness","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Converts all 12 scanned pages with no gaps; the page-11-to-page-12 handoff is preserved verbatim, and a separate page-marker output shows pages 1 through 12 present with no missing markers.","artifact_count":4,"evidence_url":"https://aidemos.com/evidence/b0876a41-66cd-455c-b38c-421968bf4e16"},{"id":"ab812bbc-7ab3-4b0c-82e1-e142cb5375fe","criterion":"visual-content-retention","criterion_name":"Visual Content Retention","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Segments a shield logo out of a single full-page raster scan and also retains Figure 1 as an image with an accurate synthesized caption; the returned logo crop is a tight 77x81px cut with legible shield text.","artifact_count":5,"evidence_url":"https://aidemos.com/evidence/ab812bbc-7ab3-4b0c-82e1-e142cb5375fe"}],"appears_in":[{"page_type":"ranking","slug":"pdf-to-markdown-apis","title":"Best AI Tools to Convert Complex PDFs into Clean Markdown with an API","url":"https://aidemos.com/best/pdf-to-markdown-apis","binding":"run"}],"same_scenario":[{"id":"347aaab3-eada-4d58-a2c1-5c5c029de742","tool":"tensorlake","tool_name":"Tensorlake","verdict":"worked","score":null,"score_total":null,"note":"Performs dedicated chart extraction on a scanned bar chart, turning the figure into structured chart content with year-by-year values, treatment labels, and the 'CUT COMPLETED' annotation."},{"id":"284ba196-edbd-492d-a9db-504bfe85f361","tool":"upstage-ai","tool_name":"Upstage AI","verdict":"worked","score":null,"score_total":null,"note":"Separately extracts Figure 3 into year-by-year values for 1972–1981 and tags it as an image asset, showing chart-specific extraction beyond plain OCR."}]}