{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P2Q7K6KZEROEYMIA33JMD3KJCG","short_pith_number":"pith:P2Q7K6KZ","schema_version":"1.0","canonical_sha256":"7ea1f57959245c4c3100ded2c1ed4911a09052efe504c3c9ef93fe1e5c8c73c4","source":{"kind":"arxiv","id":"2406.11251","version":2},"attestation_state":"computed","paper":{"title":"Unifying Multimodal Retrieval via Document Screenshot Embedding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Jimmy Lin, Minghan Li, Sheng-Chieh Lin, Wenhu Chen, Xueguang Ma","submitted_at":"2024-06-17T06:27:35Z","abstract_excerpt":"In the real world, documents are organized in different formats and varied modalities. Traditional retrieval pipelines require tailored document parsing techniques and content extraction modules to prepare input for indexing. This process is tedious, prone to errors, and has information loss. To this end, we propose Document Screenshot Embedding (DSE), a novel retrieval paradigm that regards document screenshots as a unified input format, which does not require any content extraction preprocess and preserves all the information in a document (e.g., text, image and layout). DSE leverages a larg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11251","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-06-17T06:27:35Z","cross_cats_sorted":[],"title_canon_sha256":"0c0800d7b9544cc430d41ec6a49ffdf0c085ac6978ac34d0ce57766bcb26de5e","abstract_canon_sha256":"2c27c427df71e434354365d0a73869103a0a446315601271c3b190a4f42fc3e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:43:07.913927Z","signature_b64":"WbvJ1tlgOq3Ni4UwLQzdYreU7BwnYitRzBjwOctyNtfqSo+PBF37OY5JH5c1yPbppkMvSSgaap20fQh9dOaIDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7ea1f57959245c4c3100ded2c1ed4911a09052efe504c3c9ef93fe1e5c8c73c4","last_reissued_at":"2026-07-05T09:43:07.913401Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:43:07.913401Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unifying Multimodal Retrieval via Document Screenshot Embedding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Jimmy Lin, Minghan Li, Sheng-Chieh Lin, Wenhu Chen, Xueguang Ma","submitted_at":"2024-06-17T06:27:35Z","abstract_excerpt":"In the real world, documents are organized in different formats and varied modalities. Traditional retrieval pipelines require tailored document parsing techniques and content extraction modules to prepare input for indexing. This process is tedious, prone to errors, and has information loss. To this end, we propose Document Screenshot Embedding (DSE), a novel retrieval paradigm that regards document screenshots as a unified input format, which does not require any content extraction preprocess and preserves all the information in a document (e.g., text, image and layout). DSE leverages a larg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11251","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11251/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11251","created_at":"2026-07-05T09:43:07.913462+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11251v2","created_at":"2026-07-05T09:43:07.913462+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11251","created_at":"2026-07-05T09:43:07.913462+00:00"},{"alias_kind":"pith_short_12","alias_value":"P2Q7K6KZEROE","created_at":"2026-07-05T09:43:07.913462+00:00"},{"alias_kind":"pith_short_16","alias_value":"P2Q7K6KZEROEYMIA","created_at":"2026-07-05T09:43:07.913462+00:00"},{"alias_kind":"pith_short_8","alias_value":"P2Q7K6KZ","created_at":"2026-07-05T09:43:07.913462+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05927","citing_title":"CMDR: Contextual Multimodal Document Retrieval","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2605.24530","citing_title":"Unveil: Unified Visual-Textual Integration and Distillation for Multi-modal Document Retrieval","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30027","citing_title":"DocRetriever: A Plug-and-Play Framework for Multimodal Document Retrieval with Comprehensive Benchmark","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05160","citing_title":"VLM2Vec: Training Vision-Language Models for Massive Multimodal Embedding Tasks","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2410.10594","citing_title":"VisRAG: Vision-based Retrieval-augmented Generation on Multi-modality Documents","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21262","citing_title":"CausalEmbed: Auto-Regressive Multi-Vector Generation in Latent Space for Visual Document Embedding","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07201","citing_title":"BRIDGE: Multimodal-to-Text Retrieval via Reinforcement-Learned Query Alignment","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07079","citing_title":"MARVEL: Multimodal Adaptive Reasoning-intensiVe Expand-rerank and retrievaL","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG","json":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG.json","graph_json":"https://pith.science/api/pith-number/P2Q7K6KZEROEYMIA33JMD3KJCG/graph.json","events_json":"https://pith.science/api/pith-number/P2Q7K6KZEROEYMIA33JMD3KJCG/events.json","paper":"https://pith.science/paper/P2Q7K6KZ"},"agent_actions":{"view_html":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG","download_json":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG.json","view_paper":"https://pith.science/paper/P2Q7K6KZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11251&json=true","fetch_graph":"https://pith.science/api/pith-number/P2Q7K6KZEROEYMIA33JMD3KJCG/graph.json","fetch_events":"https://pith.science/api/pith-number/P2Q7K6KZEROEYMIA33JMD3KJCG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG/action/storage_attestation","attest_author":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG/action/author_attestation","sign_citation":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG/action/citation_signature","submit_replication":"https://pith.science/pith/P2Q7K6KZEROEYMIA33JMD3KJCG/action/replication_record"}},"created_at":"2026-07-05T09:43:07.913462+00:00","updated_at":"2026-07-05T09:43:07.913462+00:00"}