{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JJCLIWZUGMXCLRQADIUF3KMX3C","short_pith_number":"pith:JJCLIWZU","schema_version":"1.0","canonical_sha256":"4a44b45b34332e25c6001a285da997d88926d93e8002d02bd06791d89d088a00","source":{"kind":"arxiv","id":"2503.11576","version":1},"attestation_state":"computed","paper":{"title":"SmolDocling: An ultra-compact vision-language model for end-to-end multi-modal document conversion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ahmed Nassar, Andres Marafioti, A. Said Gurbuz, Christoph Auer, Lucas Morin, Maksym Lysak, Matteo Omenetti, Michele Dolfi, Miquel Farr\\'e, Nikolaos Livathinos, Peter W. J. Staar, Rafael Teixeira de Lima, Yusik Kim","submitted_at":"2025-03-14T16:44:14Z","abstract_excerpt":"We introduce SmolDocling, an ultra-compact vision-language model targeting end-to-end document conversion. Our model comprehensively processes entire pages by generating DocTags, a new universal markup format that captures all page elements in their full context with location. Unlike existing approaches that rely on large foundational models, or ensemble solutions that rely on handcrafted pipelines of multiple specialized models, SmolDocling offers an end-to-end conversion for accurately capturing content, structure and spatial location of document elements in a 256M parameters vision-language"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.11576","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-14T16:44:14Z","cross_cats_sorted":[],"title_canon_sha256":"09132cfca068dd682b4a99ff14333631a45d44b529c8afa3354ced5e991856dc","abstract_canon_sha256":"ecd4e10d8d28590037259a6fdd93c674000cb8669ee17fa0d78332ca0b732abe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:35.131240Z","signature_b64":"iUK7F2AlckAtjynMtTGpEr9N9ina4hD8NEaV/ouwOJBtynLdWsEaBDYvB45KSVW+ERITznNw+H/86tx8tOb9AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a44b45b34332e25c6001a285da997d88926d93e8002d02bd06791d89d088a00","last_reissued_at":"2026-07-05T10:31:35.130518Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:35.130518Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SmolDocling: An ultra-compact vision-language model for end-to-end multi-modal document conversion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ahmed Nassar, Andres Marafioti, A. Said Gurbuz, Christoph Auer, Lucas Morin, Maksym Lysak, Matteo Omenetti, Michele Dolfi, Miquel Farr\\'e, Nikolaos Livathinos, Peter W. J. Staar, Rafael Teixeira de Lima, Yusik Kim","submitted_at":"2025-03-14T16:44:14Z","abstract_excerpt":"We introduce SmolDocling, an ultra-compact vision-language model targeting end-to-end document conversion. Our model comprehensively processes entire pages by generating DocTags, a new universal markup format that captures all page elements in their full context with location. Unlike existing approaches that rely on large foundational models, or ensemble solutions that rely on handcrafted pipelines of multiple specialized models, SmolDocling offers an end-to-end conversion for accurately capturing content, structure and spatial location of document elements in a 256M parameters vision-language"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.11576","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.11576/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.11576","created_at":"2026-07-05T10:31:35.130600+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.11576v1","created_at":"2026-07-05T10:31:35.130600+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.11576","created_at":"2026-07-05T10:31:35.130600+00:00"},{"alias_kind":"pith_short_12","alias_value":"JJCLIWZUGMXC","created_at":"2026-07-05T10:31:35.130600+00:00"},{"alias_kind":"pith_short_16","alias_value":"JJCLIWZUGMXCLRQA","created_at":"2026-07-05T10:31:35.130600+00:00"},{"alias_kind":"pith_short_8","alias_value":"JJCLIWZU","created_at":"2026-07-05T10:31:35.130600+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09788","citing_title":"POTATR: A Lightweight Image-to-Graph Model for Page-Level Table Extraction","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18025","citing_title":"TeleCom-Bench: How Far Are Large Language Models from Industrial Telecommunication Applications?","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19866","citing_title":"Structured Layout Priors for Robust Out-of-Distribution Visual Document Understanding","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22186","citing_title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05595","citing_title":"PaddleOCR 3.0 Technical Report","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12623","citing_title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00161","citing_title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18234","citing_title":"DeepSeek-OCR: Contexts Optical Compression","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C","json":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C.json","graph_json":"https://pith.science/api/pith-number/JJCLIWZUGMXCLRQADIUF3KMX3C/graph.json","events_json":"https://pith.science/api/pith-number/JJCLIWZUGMXCLRQADIUF3KMX3C/events.json","paper":"https://pith.science/paper/JJCLIWZU"},"agent_actions":{"view_html":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C","download_json":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C.json","view_paper":"https://pith.science/paper/JJCLIWZU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.11576&json=true","fetch_graph":"https://pith.science/api/pith-number/JJCLIWZUGMXCLRQADIUF3KMX3C/graph.json","fetch_events":"https://pith.science/api/pith-number/JJCLIWZUGMXCLRQADIUF3KMX3C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C/action/storage_attestation","attest_author":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C/action/author_attestation","sign_citation":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C/action/citation_signature","submit_replication":"https://pith.science/pith/JJCLIWZUGMXCLRQADIUF3KMX3C/action/replication_record"}},"created_at":"2026-07-05T10:31:35.130600+00:00","updated_at":"2026-07-05T10:31:35.130600+00:00"}