{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WVG6GYHIO3Q7BNE2U5VF24BJHA","short_pith_number":"pith:WVG6GYHI","schema_version":"1.0","canonical_sha256":"b54de360e876e1f0b49aa76a5d702938399731c91fa8d6280c02754b3ee25a97","source":{"kind":"arxiv","id":"2411.17125","version":3},"attestation_state":"computed","paper":{"title":"DOGR: Towards Versatile Visual Document Grounding and Referring","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chen Ma, Haokun Lin, Li Zhu, Shuyu Yang, Yichen Wu, Yinan Zhou, Ying Shan, Yuxin Chen, Zhongang Qi","submitted_at":"2024-11-26T05:38:34Z","abstract_excerpt":"With recent advances in Multimodal Large Language Models (MLLMs), grounding and referring capabilities have gained increasing attention for achieving detailed understanding and flexible user interaction. However, these capabilities still remain underdeveloped in visual document understanding due to the scarcity of fine-grained datasets and comprehensive benchmarks. To fill this gap, we propose the DOcument Grounding and Referring data engine (DOGR-Engine), which generates two types of high-quality fine-grained document data: (1) multi-granular parsing data to improve text localization and reco"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17125","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-26T05:38:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8c9dc587af7274dc14a2378783da5e9c71fc581e01d913e8fdd84dc5515232ee","abstract_canon_sha256":"e530bf697f503a80233ddfd78a6b196478125172bfde44b553673580a7e8c427"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:05.441059Z","signature_b64":"4HIyUkDXcimycDvsKRn2svMVay5L2nSaR5E56YvJ6V6qOBwndsMIimbtWVWobbCJ/BCZxoaTNqaNUKerAlijBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b54de360e876e1f0b49aa76a5d702938399731c91fa8d6280c02754b3ee25a97","last_reissued_at":"2026-07-05T11:49:05.440600Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:05.440600Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DOGR: Towards Versatile Visual Document Grounding and Referring","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chen Ma, Haokun Lin, Li Zhu, Shuyu Yang, Yichen Wu, Yinan Zhou, Ying Shan, Yuxin Chen, Zhongang Qi","submitted_at":"2024-11-26T05:38:34Z","abstract_excerpt":"With recent advances in Multimodal Large Language Models (MLLMs), grounding and referring capabilities have gained increasing attention for achieving detailed understanding and flexible user interaction. However, these capabilities still remain underdeveloped in visual document understanding due to the scarcity of fine-grained datasets and comprehensive benchmarks. To fill this gap, we propose the DOcument Grounding and Referring data engine (DOGR-Engine), which generates two types of high-quality fine-grained document data: (1) multi-granular parsing data to improve text localization and reco"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17125","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17125","created_at":"2026-07-05T11:49:05.440653+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17125v3","created_at":"2026-07-05T11:49:05.440653+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17125","created_at":"2026-07-05T11:49:05.440653+00:00"},{"alias_kind":"pith_short_12","alias_value":"WVG6GYHIO3Q7","created_at":"2026-07-05T11:49:05.440653+00:00"},{"alias_kind":"pith_short_16","alias_value":"WVG6GYHIO3Q7BNE2","created_at":"2026-07-05T11:49:05.440653+00:00"},{"alias_kind":"pith_short_8","alias_value":"WVG6GYHI","created_at":"2026-07-05T11:49:05.440653+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07415","citing_title":"ChartREG++: Towards Benchmarking and Improving Chart Referring Expression Grounding under Diverse referring clues and Multi-Target Referring","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07415","citing_title":"ChartREG++: Towards Benchmarking and Improving Chart Referring Expression Grounding under Diverse referring clues and Multi-Target Referring","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA","json":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA.json","graph_json":"https://pith.science/api/pith-number/WVG6GYHIO3Q7BNE2U5VF24BJHA/graph.json","events_json":"https://pith.science/api/pith-number/WVG6GYHIO3Q7BNE2U5VF24BJHA/events.json","paper":"https://pith.science/paper/WVG6GYHI"},"agent_actions":{"view_html":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA","download_json":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA.json","view_paper":"https://pith.science/paper/WVG6GYHI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17125&json=true","fetch_graph":"https://pith.science/api/pith-number/WVG6GYHIO3Q7BNE2U5VF24BJHA/graph.json","fetch_events":"https://pith.science/api/pith-number/WVG6GYHIO3Q7BNE2U5VF24BJHA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA/action/storage_attestation","attest_author":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA/action/author_attestation","sign_citation":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA/action/citation_signature","submit_replication":"https://pith.science/pith/WVG6GYHIO3Q7BNE2U5VF24BJHA/action/replication_record"}},"created_at":"2026-07-05T11:49:05.440653+00:00","updated_at":"2026-07-05T11:49:05.440653+00:00"}