{"observation":{"id":"4179e969-c45b-443e-b8cf-19142cd6f74b","tool":"firecrawl","tool_name":"Firecrawl","criterion":"noise-filtering","criterion_name":"Noise Filtering","criterion_definition":"How well the tool strips boilerplate, ads, navigation, comments, and other clutter from a static page.","criterion_evidence_type":"transformation","criterion_rank_role":"decisive","criterion_rank_role_reason":"Removing ads, nav, comments, and other clutter is fundamental to producing clean Markdown or structured data. (3 of 3 judges)","scenario":"glassdoor-software-engineer-jobs-behind-sign-in-modal","scenario_name":"Glassdoor software engineer jobs behind sign-in modal","group_tag":"web-scraping-benchmark","scenario_description":"A Glassdoor jobs listing page protected by Cloudflare and a sign-in/interstitial overlay, used to test proxy evasion, anti-bot handling, and the ability to dismiss blocking modal UI before extracting listings.","modality":"mixed","input_text":"https://www.glassdoor.com/Job/software-engineer-jobs-SRCH_KO0,17.htm — Dismiss any immediate sign-in or signup modal overlays that block the view. Once cleared, extract the top 5 job listings, including job title, company name, location, and the short summary snippet.","input_artifact_refs":[],"stresses":[],"verdict":"failed","score":null,"score_total":null,"note":"Fails to strip boilerplate, retaining the full primary navigation tree, historical sidebar components, thousands of user review nodes, and the footer block.","evidence_state":"observed","source":"first-party","artifacts":[],"run_id":"06e1dbd6-5518-4af8-aa1a-735259a75b4f","study_title":"Scrape Web Pages Into Clean Markdown or Structured Data Using AI","study_kind":"backfill","research_task":"86b9jm3a3","tested_at":"2026-06-23T04:32:57.304000+00:00","completeness":"input-only","input":{"state":"text","text":"https://www.glassdoor.com/Job/software-engineer-jobs-SRCH_KO0,17.htm — Dismiss any immediate sign-in or signup modal overlays that block the view. Once cleared, extract the top 5 job listings, including job title, company name, location, and the short summary snippet.","files":[],"modality":"mixed","stresses":[]},"tool_page_slug":"firecrawl","tool_url":"https://aidemos.com/tools/firecrawl","permalink":"https://aidemos.com/evidence/4179e969-c45b-443e-b8cf-19142cd6f74b","api_url":"https://ai.aidemos.com/v1/observations/4179e969-c45b-443e-b8cf-19142cd6f74b"},"peers":[{"id":"1c2ec4aa-ed8e-4021-8ef3-56d35ce08689","tool":"jina-ai-reader","tool_name":"Jina AI Reader","verdict":"mixed","score":null,"score_total":null,"note":"It can recover the page text layer, but the extraction still leaves job data interleaved with French and German translation strings and header redirect text, so heavy post-processing cleanup is still required.","artifact_count":0,"thumbnail":null,"evidence_url":"https://aidemos.com/evidence/1c2ec4aa-ed8e-4021-8ef3-56d35ce08689"},{"id":"f027509e-7bee-4fa6-8580-a72e8c2950e4","tool":"spider","tool_name":"Spider","verdict":"failed","score":null,"score_total":null,"note":"The scraper fails to strip static boilerplate from cluttered pages: the markdown included the global header navigation, social-sharing URLs, cookie-choice notices, and user reviews instead of isolating only the core recipe content.","artifact_count":0,"thumbnail":null,"evidence_url":"https://aidemos.com/evidence/f027509e-7bee-4fa6-8580-a72e8c2950e4"}],"other_criteria":[{"id":"f74f2dc5-6024-446d-90db-244bd5c2394f","criterion":"input-handling","criterion_name":"Input Handling","rank_role":"context","verdict":"worked","score":null,"score_total":null,"note":"Accepted the protected Glassdoor jobs URL and began extraction behind the edge layer without access or parsing errors.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/f74f2dc5-6024-446d-90db-244bd5c2394f"},{"id":"0a0ad5f0-a8cd-4625-83c4-b4bdbef9390a","criterion":"interaction-stability","criterion_name":"Interaction Stability","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Survives Cloudflare-style edge/proxy behavior and still returns text structures from behind the firewall.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/0a0ad5f0-a8cd-4625-83c4-b4bdbef9390a"},{"id":"578ce21d-ead2-4717-a60a-d00722e7a805","criterion":"output-quality","criterion_name":"Output Quality","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Preserves Markdown structure and core content cleanly, including headings, the ingredients table, and the step-by-step workflow, with excellent textual fidelity.","artifact_count":0,"evidence_url":"https://aidemos.com/evidence/578ce21d-ead2-4717-a60a-d00722e7a805"},{"id":"c891d64f-05c0-4ce9-8b5c-8cd2c291ec80","criterion":"output-quality","criterion_name":"Output Quality","rank_role":"decisive","verdict":"mixed","score":null,"score_total":null,"note":"Captures the key product fields, but the output quality is degraded by raw backend code artifacts and raw media attachment matrices mixed into the document.","artifact_count":0,"evidence_url":"https://aidemos.com/evidence/c891d64f-05c0-4ce9-8b5c-8cd2c291ec80"},{"id":"9ed0263e-7874-4c44-a0d2-c55dd52cca1c","criterion":"output-quality","criterion_name":"Output Quality","rank_role":"decisive","verdict":"mixed","score":null,"score_total":null,"note":"Extracts the core job-listing content, but breaks the text structure with navigation buttons, search filter blocks, and internal page links.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/9ed0263e-7874-4c44-a0d2-c55dd52cca1c"},{"id":"becc8618-b56d-4b69-9774-2f69f3610041","criterion":"proxy-evasion","criterion_name":"Proxy Evasion","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Bypasses Glassdoor's Cloudflare-style perimeter defenses and recovers the target job listing content, including active software-engineering listings, corporate profile names, salary estimates, and technical-skill arrays.","artifact_count":0,"evidence_url":"https://aidemos.com/evidence/becc8618-b56d-4b69-9774-2f69f3610041"},{"id":"3d6a4fa2-e8e0-4a86-b2f5-866746cfeb98","criterion":"proxy-evasion","criterion_name":"Proxy Evasion","rank_role":"decisive","verdict":"mixed","score":null,"score_total":null,"note":"Returns the protected job page in a noisy flattened form, interleaving the target content with navigation buttons, search filter blocks, and internal page links instead of a cleanly isolated listing block.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/3d6a4fa2-e8e0-4a86-b2f5-866746cfeb98"},{"id":"1c24aecb-5c9f-48fc-aed0-261254b00f5e","criterion":"proxy-evasion","criterion_name":"Proxy Evasion","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"Gets past Cloudflare-protected interstitial defenses and returns the underlying job-listing content from behind the barrier.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/1c24aecb-5c9f-48fc-aed0-261254b00f5e"},{"id":"4a5a4574-2020-475c-9c89-24c4e72996c2","criterion":"schema-extraction-integrity","criterion_name":"Schema Extraction Integrity","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"The extractor returned the core job-listing payload accurately, including active software engineering listings, corporate profile names, salary estimates, and required technical skill arrays.","artifact_count":0,"evidence_url":"https://aidemos.com/evidence/4a5a4574-2020-475c-9c89-24c4e72996c2"},{"id":"b0430dd1-be01-4f0f-8243-fc559ddcd2e5","criterion":"schema-extraction-integrity","criterion_name":"Schema Extraction Integrity","rank_role":"decisive","verdict":"worked","score":null,"score_total":null,"note":"The tool preserved the requested recipe structure with high textual fidelity, including the ingredients table and step-by-step workflow, and retained hyperlink routing definitions accurately.","artifact_count":0,"evidence_url":"https://aidemos.com/evidence/b0430dd1-be01-4f0f-8243-fc559ddcd2e5"},{"id":"be5170f7-974f-47c9-b270-65724272315c","criterion":"visual-spatial-awareness","criterion_name":"Visual Spatial Awareness","rank_role":"decisive","verdict":"failed","score":null,"score_total":null,"note":"Fails to separate the target job region from surrounding page chrome, returning global layout blocks immediately after the main content.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/be5170f7-974f-47c9-b270-65724272315c"},{"id":"e050f6e3-c13c-4afb-94ce-dd54f6708cc8","criterion":"visual-spatial-awareness","criterion_name":"Visual Spatial Awareness","rank_role":"decisive","verdict":"failed","score":null,"score_total":null,"note":"The parser did not isolate primary content from boilerplate: it flattened the full navigation tree, sidebar modules, thousands of review nodes, and the footer into the same markdown block as the article text.","artifact_count":0,"evidence_url":"https://aidemos.com/evidence/e050f6e3-c13c-4afb-94ce-dd54f6708cc8"},{"id":"b720b17e-c660-4ca4-b647-590fb4e1b465","criterion":"visual-spatial-awareness","criterion_name":"Visual Spatial Awareness","rank_role":"decisive","verdict":"failed","score":null,"score_total":null,"note":"The page text was not structurally separated from UI noise, with navigation buttons, search filter blocks, internal links, and login fields interleaved directly around the job listing content.","artifact_count":0,"evidence_url":"https://aidemos.com/evidence/b720b17e-c660-4ca4-b647-590fb4e1b465"},{"id":"6e67d0a2-37f7-4b8c-aaab-7571c730a63c","criterion":"visual-spatial-awareness","criterion_name":"Visual Spatial Awareness","rank_role":"decisive","verdict":"failed","score":null,"score_total":null,"note":"Fails to separate the job content from surrounding page chrome; the output still carries global layout blocks and login fields alongside the listings.","artifact_count":1,"evidence_url":"https://aidemos.com/evidence/6e67d0a2-37f7-4b8c-aaab-7571c730a63c"}],"appears_in":[{"page_type":"ranking","slug":"ai-web-scraping-tools","title":"Best AI Tools to Scrape Web Pages Into Clean Markdown or Structured Data","url":"https://aidemos.com/best/ai-web-scraping-tools","binding":"run"}],"same_scenario":[{"id":"1c2ec4aa-ed8e-4021-8ef3-56d35ce08689","tool":"jina-ai-reader","tool_name":"Jina AI Reader","verdict":"mixed","score":null,"score_total":null,"note":"It can recover the page text layer, but the extraction still leaves job data interleaved with French and German translation strings and header redirect text, so heavy post-processing cleanup is still required."},{"id":"f027509e-7bee-4fa6-8580-a72e8c2950e4","tool":"spider","tool_name":"Spider","verdict":"failed","score":null,"score_total":null,"note":"The scraper fails to strip static boilerplate from cluttered pages: the markdown included the global header navigation, social-sharing URLs, cookie-choice notices, and user reviews instead of isolating only the core recipe content."}]}