{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UOE2QQYWW6BBF7GICFBMYTWIMX","short_pith_number":"pith:UOE2QQYW","schema_version":"1.0","canonical_sha256":"a389a84316b78212fcc81142cc4ec865eaf908e58ab653315ac5f150baf9a208","source":{"kind":"arxiv","id":"2504.07089","version":3},"attestation_state":"computed","paper":{"title":"OmniCaptioner: One Captioner to Rule Them All","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Botian Shi, Bo Zhang, Dongyang Liu, Jiakang Yuan, Lei Bai, Le Zhuo, Licheng Wen, Peng Gao, Qi Qin, Shitian Zhao, Shufei Zhang, Tao Chen, Tianshuo Peng, Xiangchao Yan, Xin Li, Xinyue Li, Yiting Lu, Yuewen Cao, Zhen Li, Zhibo Chen","submitted_at":"2025-04-09T17:58:58Z","abstract_excerpt":"We propose OmniCaptioner, a versatile visual captioning framework for generating fine-grained textual descriptions across a wide variety of visual domains. Unlike prior methods limited to specific image types (e.g., natural images or geometric visuals), our framework provides a unified solution for captioning natural images, visual text (e.g., posters, UIs, textbooks), and structured visuals (e.g., documents, tables, charts). By converting low-level pixel information into semantically rich textual representations, our framework bridges the gap between visual and textual modalities. Our results"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07089","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-09T17:58:58Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"fcc148ccf9e7d5c19e77a4f4c45ba790d23fda042887d7de6ffc5047c66725a8","abstract_canon_sha256":"ce14449541b249f3afdfa392145d2b883e04ed025c95a69d81e99b9c370b973e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:06.285680Z","signature_b64":"niALZfAXG93WieAixiK6Eze38PxsXnXucUwoZxHDXjq8THMgeV7e+yVg3V7EiTCsHo20F+4bdJhWEaPxC/7XDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a389a84316b78212fcc81142cc4ec865eaf908e58ab653315ac5f150baf9a208","last_reissued_at":"2026-07-05T11:14:06.285153Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:06.285153Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OmniCaptioner: One Captioner to Rule Them All","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Botian Shi, Bo Zhang, Dongyang Liu, Jiakang Yuan, Lei Bai, Le Zhuo, Licheng Wen, Peng Gao, Qi Qin, Shitian Zhao, Shufei Zhang, Tao Chen, Tianshuo Peng, Xiangchao Yan, Xin Li, Xinyue Li, Yiting Lu, Yuewen Cao, Zhen Li, Zhibo Chen","submitted_at":"2025-04-09T17:58:58Z","abstract_excerpt":"We propose OmniCaptioner, a versatile visual captioning framework for generating fine-grained textual descriptions across a wide variety of visual domains. Unlike prior methods limited to specific image types (e.g., natural images or geometric visuals), our framework provides a unified solution for captioning natural images, visual text (e.g., posters, UIs, textbooks), and structured visuals (e.g., documents, tables, charts). By converting low-level pixel information into semantically rich textual representations, our framework bridges the gap between visual and textual modalities. Our results"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07089","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07089/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07089","created_at":"2026-07-05T11:14:06.285210+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07089v3","created_at":"2026-07-05T11:14:06.285210+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07089","created_at":"2026-07-05T11:14:06.285210+00:00"},{"alias_kind":"pith_short_12","alias_value":"UOE2QQYWW6BB","created_at":"2026-07-05T11:14:06.285210+00:00"},{"alias_kind":"pith_short_16","alias_value":"UOE2QQYWW6BBF7GI","created_at":"2026-07-05T11:14:06.285210+00:00"},{"alias_kind":"pith_short_8","alias_value":"UOE2QQYW","created_at":"2026-07-05T11:14:06.285210+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05716","citing_title":"Scene Graph Thinking: Reinforcing Structured Visual Reasoning for Multimodal Large Language Models","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22699","citing_title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05623","citing_title":"DetailVerifyBench: A Benchmark for Dense Hallucination Localization in Long Image Captions","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04503","citing_title":"DiffCap-Bench: A Comprehensive, Challenging, Robust Benchmark for Image Difference Captioning","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX","json":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX.json","graph_json":"https://pith.science/api/pith-number/UOE2QQYWW6BBF7GICFBMYTWIMX/graph.json","events_json":"https://pith.science/api/pith-number/UOE2QQYWW6BBF7GICFBMYTWIMX/events.json","paper":"https://pith.science/paper/UOE2QQYW"},"agent_actions":{"view_html":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX","download_json":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX.json","view_paper":"https://pith.science/paper/UOE2QQYW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07089&json=true","fetch_graph":"https://pith.science/api/pith-number/UOE2QQYWW6BBF7GICFBMYTWIMX/graph.json","fetch_events":"https://pith.science/api/pith-number/UOE2QQYWW6BBF7GICFBMYTWIMX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX/action/storage_attestation","attest_author":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX/action/author_attestation","sign_citation":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX/action/citation_signature","submit_replication":"https://pith.science/pith/UOE2QQYWW6BBF7GICFBMYTWIMX/action/replication_record"}},"created_at":"2026-07-05T11:14:06.285210+00:00","updated_at":"2026-07-05T11:14:06.285210+00:00"}