{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:GHX4G5UAOD5TIRBWQAW6BQYW2Q","short_pith_number":"pith:GHX4G5UA","schema_version":"1.0","canonical_sha256":"31efc3768070fb344436802de0c316d40b180076a97a0fa0cfb427887112d522","source":{"kind":"arxiv","id":"2608.08261","version":1},"attestation_state":"computed","paper":{"title":"Scout: Scalable Document Extraction via Data Similarity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Aditya G. Parameswaran, Chiyu Hao, Shreya Shankar, Yiming Lin","submitted_at":"2026-08-08T17:47:31Z","abstract_excerpt":"Extracting values from large document collections powers data analysis across many domains. Frontier LLMs extract such values accurately, but processing an\n  entire collection with one is prohibitively costly. Yet this cost is largely avoidable: real-world collections exhibit rich similarity, so for the same query\n  over similar documents, the answer tends to recur in similar locations; an LLM need only read that small span, not the whole document. Prior methods that\n  exploit this similarity fall short: they either assume a rigid document structure, or assume the answer is a set of substrings"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.08261","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DB","submitted_at":"2026-08-08T17:47:31Z","cross_cats_sorted":[],"title_canon_sha256":"a4858b6bd854b5b39c46edb694f2b4f2e5caf424c69f3e29de6881c9fd281cc5","abstract_canon_sha256":"447542a3d200332139cb7730a39aabf7267651d89e225ac956c5f081328fa12d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T01:22:09.244601Z","signature_b64":"YSevYZN3ZzdDLHexbyqCkpvKiL0SgeB9Wf/4AionKQ/U3wLR74zdKwdU4jx+CyV5CLWCn02TevcG2wm8gT4nBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31efc3768070fb344436802de0c316d40b180076a97a0fa0cfb427887112d522","last_reissued_at":"2026-08-11T01:22:09.242240Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T01:22:09.242240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scout: Scalable Document Extraction via Data Similarity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Aditya G. Parameswaran, Chiyu Hao, Shreya Shankar, Yiming Lin","submitted_at":"2026-08-08T17:47:31Z","abstract_excerpt":"Extracting values from large document collections powers data analysis across many domains. Frontier LLMs extract such values accurately, but processing an\n  entire collection with one is prohibitively costly. Yet this cost is largely avoidable: real-world collections exhibit rich similarity, so for the same query\n  over similar documents, the answer tends to recur in similar locations; an LLM need only read that small span, not the whole document. Prior methods that\n  exploit this similarity fall short: they either assume a rigid document structure, or assume the answer is a set of substrings"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.08261","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.08261/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.08261","created_at":"2026-08-11T01:22:09.242966+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.08261v1","created_at":"2026-08-11T01:22:09.242966+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.08261","created_at":"2026-08-11T01:22:09.242966+00:00"},{"alias_kind":"pith_short_12","alias_value":"GHX4G5UAOD5T","created_at":"2026-08-11T01:22:09.242966+00:00"},{"alias_kind":"pith_short_16","alias_value":"GHX4G5UAOD5TIRBW","created_at":"2026-08-11T01:22:09.242966+00:00"},{"alias_kind":"pith_short_8","alias_value":"GHX4G5UA","created_at":"2026-08-11T01:22:09.242966+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q","json":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q.json","graph_json":"https://pith.science/api/pith-number/GHX4G5UAOD5TIRBWQAW6BQYW2Q/graph.json","events_json":"https://pith.science/api/pith-number/GHX4G5UAOD5TIRBWQAW6BQYW2Q/events.json","paper":"https://pith.science/paper/GHX4G5UA"},"agent_actions":{"view_html":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q","download_json":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q.json","view_paper":"https://pith.science/paper/GHX4G5UA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.08261&json=true","fetch_graph":"https://pith.science/api/pith-number/GHX4G5UAOD5TIRBWQAW6BQYW2Q/graph.json","fetch_events":"https://pith.science/api/pith-number/GHX4G5UAOD5TIRBWQAW6BQYW2Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q/action/storage_attestation","attest_author":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q/action/author_attestation","sign_citation":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q/action/citation_signature","submit_replication":"https://pith.science/pith/GHX4G5UAOD5TIRBWQAW6BQYW2Q/action/replication_record"}},"created_at":"2026-08-11T01:22:09.242966+00:00","updated_at":"2026-08-11T01:22:09.242966+00:00"}