{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TVEIOR7XAJ4QSVDADQ7SH3ES7A","short_pith_number":"pith:TVEIOR7X","schema_version":"1.0","canonical_sha256":"9d488747f702790954601c3f23ec92f83fd9788d8689bab462bb0589f3160bea","source":{"kind":"arxiv","id":"2407.08303","version":2},"attestation_state":"computed","paper":{"title":"DenseFusion-1M: Merging Vision Experts for Comprehensive Multimodal Perception","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Haiwen Diao, Ling-Yu Duan, Xiaotong Li, Xinlong Wang, Yueze Wang","submitted_at":"2024-07-11T08:48:06Z","abstract_excerpt":"Existing Multimodal Large Language Models (MLLMs) increasingly emphasize complex understanding of various visual elements, including multiple objects, text information, and spatial relations. Their development for comprehensive visual perception hinges on the availability of high-quality image-text datasets that offer diverse visual elements and throughout image descriptions. However, the scarcity of such hyper-detailed datasets currently hinders progress within the MLLM community. The bottleneck stems from the limited perceptual capabilities of current caption engines, which fall short in pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08303","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-11T08:48:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d6c8b9629d1dcba160e191f5558031c4b1c1c41ddd611d94b34b1176dc6ae6d1","abstract_canon_sha256":"eaaddef3a05b3b76dfb5d8e3623b05129030c1ee3ca5a071a3c6a90757e51beb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:24.932111Z","signature_b64":"q+DsMN/OHF4RTRgIkTG8uWKkiNu+XMZ9kLKnjuqX8Lsbhxh+2G9v/Isc6s5RjUaMnetzP6nWROoCilawfqVGAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d488747f702790954601c3f23ec92f83fd9788d8689bab462bb0589f3160bea","last_reissued_at":"2026-07-05T09:39:24.931636Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:24.931636Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DenseFusion-1M: Merging Vision Experts for Comprehensive Multimodal Perception","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Haiwen Diao, Ling-Yu Duan, Xiaotong Li, Xinlong Wang, Yueze Wang","submitted_at":"2024-07-11T08:48:06Z","abstract_excerpt":"Existing Multimodal Large Language Models (MLLMs) increasingly emphasize complex understanding of various visual elements, including multiple objects, text information, and spatial relations. Their development for comprehensive visual perception hinges on the availability of high-quality image-text datasets that offer diverse visual elements and throughout image descriptions. However, the scarcity of such hyper-detailed datasets currently hinders progress within the MLLM community. The bottleneck stems from the limited perceptual capabilities of current caption engines, which fall short in pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08303","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08303/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08303","created_at":"2026-07-05T09:39:24.931690+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08303v2","created_at":"2026-07-05T09:39:24.931690+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08303","created_at":"2026-07-05T09:39:24.931690+00:00"},{"alias_kind":"pith_short_12","alias_value":"TVEIOR7XAJ4Q","created_at":"2026-07-05T09:39:24.931690+00:00"},{"alias_kind":"pith_short_16","alias_value":"TVEIOR7XAJ4QSVDA","created_at":"2026-07-05T09:39:24.931690+00:00"},{"alias_kind":"pith_short_8","alias_value":"TVEIOR7X","created_at":"2026-07-05T09:39:24.931690+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09393","citing_title":"CapRL++: Unified Reinforcement Learning with Verifiable Rewards for Dense Image and Video Captioning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18871","citing_title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13848","citing_title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15564","citing_title":"Show-o2: Improved Native Unified Multimodal Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18869","citing_title":"Emu3: Next-Token Prediction is All You Need","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10302","citing_title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13106","citing_title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A","json":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A.json","graph_json":"https://pith.science/api/pith-number/TVEIOR7XAJ4QSVDADQ7SH3ES7A/graph.json","events_json":"https://pith.science/api/pith-number/TVEIOR7XAJ4QSVDADQ7SH3ES7A/events.json","paper":"https://pith.science/paper/TVEIOR7X"},"agent_actions":{"view_html":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A","download_json":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A.json","view_paper":"https://pith.science/paper/TVEIOR7X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08303&json=true","fetch_graph":"https://pith.science/api/pith-number/TVEIOR7XAJ4QSVDADQ7SH3ES7A/graph.json","fetch_events":"https://pith.science/api/pith-number/TVEIOR7XAJ4QSVDADQ7SH3ES7A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A/action/storage_attestation","attest_author":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A/action/author_attestation","sign_citation":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A/action/citation_signature","submit_replication":"https://pith.science/pith/TVEIOR7XAJ4QSVDADQ7SH3ES7A/action/replication_record"}},"created_at":"2026-07-05T09:39:24.931690+00:00","updated_at":"2026-07-05T09:39:24.931690+00:00"}