{"observation":{"id":"32fab15d-0885-4287-94a3-67e0d1d00611","tool":"adobe-api","tool_name":"Adobe API","criterion":"complex-document-handling","criterion_name":"Complex Document Handling","criterion_definition":"Maintains quality across long, multi-section, and mixed-content documents without degradation.","criterion_evidence_type":"transformation","criterion_rank_role":"decisive","criterion_rank_role_reason":"The subject explicitly says complex PDF, so sustained quality across long, mixed-content documents is central to the tool’s ability to do the job. (3 of 3 judges)","scenario":"scanned-research-paper","scenario_name":"Scanned Research Paper","group_tag":"scanned-research-paper","scenario_description":"An image-only scanned research paper used to stress OCR and layout recovery in a multi-column academic document with figures, charts, tables, captions, and references.","modality":"pdf","input_text":null,"input_artifact_refs":[{"alt":null,"url":"https://d3epheqghktydj.cloudfront.net/convert-a-complex-pdf-into-clean-markdow-scanned-research-pdf-7b86de49784d.pdf","role":"input","filename":"Scanned Research PDF.pdf"}],"stresses":["OCR on scanned pages","Multi-column reading order","Figure and chart handling","Table reconstruction from scans","Caption association","Reference extraction","Overall document structure retention"],"verdict":"struggled","score":null,"score_total":null,"note":"Requires splitting an oversized scanned paper into two PDFs before processing, because the web interface rejects uploads over 1 MB.","evidence_state":"verified","source":null,"artifacts":[{"url":"https://cdn.futuresmart.ai/public/aidemos/790f9221306f4bd98ec30e9c44908862.mp4?v=1","role":"context","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/c2947de05ef647539af8d63a9a2b27c4.mp4?v=1","role":"context","alt":null},{"url":"https://d3epheqghktydj.cloudfront.net/research-media-scanned-research-pdf-pages-1-to-6-output-851ae2ab3972.md","role":"output","alt":null},{"url":"https://d3epheqghktydj.cloudfront.net/research-media-scanned-research-pdf-pages-7-to-12-outpu-bac957952556.md","role":"output","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/ca474482f7c9449a9604752fffe26cd7.pdf?v=1","role":"input","alt":null},{"url":"https://cdn.futuresmart.ai/public/aidemos/a7dec7a1fa80425fa8453eaa1632e7d8.pdf?v=1","role":"input","alt":null}],"run_id":"6e3160de-fe46-4b45-b071-72560b5c5d0e","study_title":"Convert a Complex PDF into Clean Markdown with an API","study_kind":"generation","research_task":"86b9h7t37","tested_at":null,"completeness":"input-and-output","input":{"state":"files","text":null,"files":[{"url":"https://d3epheqghktydj.cloudfront.net/convert-a-complex-pdf-into-clean-markdow-scanned-research-pdf-7b86de49784d.pdf","filename":"Scanned Research PDF.pdf","alt":"Scanned Research Paper","role":"input"}],"modality":"pdf","stresses":["OCR on scanned pages","Multi-column reading order","Figure and chart handling","Table reconstruction from scans","Caption association","Reference extraction","Overall document structure retention"]},"tool_page_slug":"adobe-api","tool_url":"https://aidemos.com/tools/adobe-api","permalink":"https://aidemos.com/evidence/32fab15d-0885-4287-94a3-67e0d1d00611","api_url":"https://ai.aidemos.com/v1/observations/32fab15d-0885-4287-94a3-67e0d1d00611"},"peers":[{"id":"c8ea9121-f878-4e78-bf13-3ba3b4b6f0a5","tool":"extend-ai","tool_name":"Extend AI","verdict":"worked","score":null,"score_total":null,"note":"Handles the scanned research paper end-to-end, including multi-column prose, tables, charts, and handwritten marginalia, while still producing parsed markdown.","artifact_count":3,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/45e3533a31c246b29e0ec6aaa98438e4.pdf?v=1","evidence_url":"https://aidemos.com/evidence/c8ea9121-f878-4e78-bf13-3ba3b4b6f0a5"},{"id":"9f0d06be-7ecb-4732-933c-ea6d52a1cc16","tool":"llamaparse","tool_name":"LlamaParse","verdict":"worked","score":null,"score_total":null,"note":"Processes a 12-page scanned paper end-to-end and reaches SUCCESS after extracting the page content and figure/table outputs.","artifact_count":3,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/9848fddc3df54deda1f31e5ae8d86f56.mp4?v=1","evidence_url":"https://aidemos.com/evidence/9f0d06be-7ecb-4732-933c-ea6d52a1cc16"},{"id":"d572c22d-d6ae-4a81-b848-53c180d00575","tool":"mistral-ai","tool_name":"Mistral AI","verdict":"worked","score":null,"score_total":null,"note":"The tool processes a scanned multi-column research paper end-to-end and returns OCR text, tables, and embedded chart assets in page-wise markdown output.","artifact_count":4,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/f84ba080d3254487844fc0a3f51f1873.mp4?v=1","evidence_url":"https://aidemos.com/evidence/d572c22d-d6ae-4a81-b848-53c180d00575"},{"id":"b50f09d7-7cee-4268-8441-50b46ddeb000","tool":"nutrient-io","tool_name":"Nutrient.io","verdict":"worked","score":null,"score_total":null,"note":"Processes an image-only scanned research paper end to end and returns a parsed markdown output plus preview, showing it can handle a multi-page scanned document.","artifact_count":4,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/45e3533a31c246b29e0ec6aaa98438e4.pdf?v=1","evidence_url":"https://aidemos.com/evidence/b50f09d7-7cee-4268-8441-50b46ddeb000"},{"id":"3bec7ed4-9b8e-44f3-b548-b76f0215b34c","tool":"reducto","tool_name":"Reducto","verdict":"struggled","score":null,"score_total":null,"note":"Handles the 12-page scan in 10.6 seconds with no truncation, but the densest 17-column table has unreadable stretches with injected glyphs and run-on numbers.","artifact_count":2,"thumbnail":"https://d3epheqghktydj.cloudfront.net/research-media-reducto-input3-scannedpaper-output-6dfdac76e722.md","evidence_url":"https://aidemos.com/evidence/3bec7ed4-9b8e-44f3-b548-b76f0215b34c"},{"id":"ea443095-bb7a-45ec-92ca-6ff0afefe8e4","tool":"tensorlake","tool_name":"Tensorlake","verdict":"mixed","score":null,"score_total":null,"note":"On the scanned research paper, section flow and chart extraction work, but hierarchical tables degrade, so mixed-content handling is uneven rather than consistently robust.","artifact_count":8,"thumbnail":"https://cdn.futuresmart.ai/public/aidemos/18b2277b09fa42e3ae7a325fba02b272.png?v=1","evidence_url":"https://aidemos.com/evidence/ea443095-bb7a-45ec-92ca-6ff0afefe8e4"}],"other_criteria":[{"id":"24968ab8-597c-4c69-9ec3-a0c216320ebc","criterion":"reading-order-structure","criterion_name":"Reading Order & Structure","rank_role":"decisive","verdict":"failed","score":null,"score_total":null,"note":"Dumps the scanned title page as a dense OCR block without section boundaries or other structural cues, so the document hierarchy is lost.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/24968ab8-597c-4c69-9ec3-a0c216320ebc"},{"id":"82c91005-2538-47cc-812c-da8ff0c240c3","criterion":"table-preservation","criterion_name":"Table Preservation","rank_role":"decisive","verdict":"failed","score":null,"score_total":null,"note":"Breaks a grouped-column table when intervening text appears, fragmenting the 1979–1981 layout and corrupting the extracted alignment.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/82c91005-2538-47cc-812c-da8ff0c240c3"},{"id":"1dbde32e-95eb-4904-8433-272d8c5e1fdb","criterion":"table-preservation","criterion_name":"Table Preservation","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Preserves a multi-column scientific table of tree treatments and before/after diameters, keeping the treatment rows and inch/cm subcolumns readable.","artifact_count":2,"evidence_url":"https://aidemos.com/evidence/1dbde32e-95eb-4904-8433-272d8c5e1fdb"},{"id":"ec57bfbf-d0f3-4011-803c-9f42ea53397b","criterion":"text-ocr-completeness","criterion_name":"Text & OCR Completeness","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Recovers the visible title, abstract, keywords, and opening paragraphs from a scanned USDA forestry report as dense OCR text.","artifact_count":3,"evidence_url":"https://aidemos.com/evidence/ec57bfbf-d0f3-4011-803c-9f42ea53397b"},{"id":"aaf02083-a041-4bdf-b6af-90778cdde03b","criterion":"visual-content-retention","criterion_name":"Visual Content Retention","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Keeps chart artwork embedded in the extracted page, placing the residual basal-area figure in situ beneath the extracted table text.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/aaf02083-a041-4bdf-b6af-90778cdde03b"}],"appears_in":[{"page_type":"ranking","slug":"pdf-to-markdown-apis","title":"Best AI Tools to Convert Complex PDFs into Clean Markdown with an API","url":"https://aidemos.com/best/pdf-to-markdown-apis","binding":"run"}],"same_scenario":[{"id":"c8ea9121-f878-4e78-bf13-3ba3b4b6f0a5","tool":"extend-ai","tool_name":"Extend AI","verdict":"worked","score":null,"score_total":null,"note":"Handles the scanned research paper end-to-end, including multi-column prose, tables, charts, and handwritten marginalia, while still producing parsed markdown."},{"id":"9f0d06be-7ecb-4732-933c-ea6d52a1cc16","tool":"llamaparse","tool_name":"LlamaParse","verdict":"worked","score":null,"score_total":null,"note":"Processes a 12-page scanned paper end-to-end and reaches SUCCESS after extracting the page content and figure/table outputs."},{"id":"d572c22d-d6ae-4a81-b848-53c180d00575","tool":"mistral-ai","tool_name":"Mistral AI","verdict":"worked","score":null,"score_total":null,"note":"The tool processes a scanned multi-column research paper end-to-end and returns OCR text, tables, and embedded chart assets in page-wise markdown output."},{"id":"b50f09d7-7cee-4268-8441-50b46ddeb000","tool":"nutrient-io","tool_name":"Nutrient.io","verdict":"worked","score":null,"score_total":null,"note":"Processes an image-only scanned research paper end to end and returns a parsed markdown output plus preview, showing it can handle a multi-page scanned document."},{"id":"3bec7ed4-9b8e-44f3-b548-b76f0215b34c","tool":"reducto","tool_name":"Reducto","verdict":"struggled","score":null,"score_total":null,"note":"Handles the 12-page scan in 10.6 seconds with no truncation, but the densest 17-column table has unreadable stretches with injected glyphs and run-on numbers."},{"id":"ea443095-bb7a-45ec-92ca-6ff0afefe8e4","tool":"tensorlake","tool_name":"Tensorlake","verdict":"mixed","score":null,"score_total":null,"note":"On the scanned research paper, section flow and chart extraction work, but hierarchical tables degrade, so mixed-content handling is uneven rather than consistently robust."}]}