{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JC6MAHAJI3Y5P5HZAVOM5Q2IIU","short_pith_number":"pith:JC6MAHAJ","schema_version":"1.0","canonical_sha256":"48bcc01c0946f1d7f4f9055ccec3484517a0f791bdf337af996ce25eb996ea70","source":{"kind":"arxiv","id":"2509.00723","version":1},"attestation_state":"computed","paper":{"title":"OmniDPO: A Preference Optimization Framework to Address Omni-Modal Hallucination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.AI","authors_text":"Chao Sun, Guanyu Zhou, Junzhe Chen, Lijie Wen, Rongzhou Zhang, Shiyu Huang, Tianshu Zhang, Xuming Hu, Yuwei Niu","submitted_at":"2025-08-31T07:19:32Z","abstract_excerpt":"Recently, Omni-modal large language models (OLLMs) have sparked a new wave of research, achieving impressive results in tasks such as audio-video understanding and real-time environment perception. However, hallucination issues still persist. Similar to the bimodal setting, the priors from the text modality tend to dominate, leading OLLMs to rely more heavily on textual cues while neglecting visual and audio information. In addition, fully multimodal scenarios introduce new challenges. Most existing models align visual or auditory modalities with text independently during training, while ignor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.00723","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-08-31T07:19:32Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"293e977f082dd15399a3f9a1d8b903523f602fba7b4006853c5b7a5bbfc18909","abstract_canon_sha256":"02f8bc43a0b611aac57c40cad389ef45b669cddc0713e4f48c8835c1ddf587e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:21.780765Z","signature_b64":"FNA/EHRkSw4P8ckg+TgrdUeOzdbTBnwL//AfIf9WCQIEXSp/5vljmriSniGUVFony06b0kRhNGfQXK/7Q9K5Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48bcc01c0946f1d7f4f9055ccec3484517a0f791bdf337af996ce25eb996ea70","last_reissued_at":"2026-07-05T12:02:21.779356Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:21.779356Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OmniDPO: A Preference Optimization Framework to Address Omni-Modal Hallucination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.AI","authors_text":"Chao Sun, Guanyu Zhou, Junzhe Chen, Lijie Wen, Rongzhou Zhang, Shiyu Huang, Tianshu Zhang, Xuming Hu, Yuwei Niu","submitted_at":"2025-08-31T07:19:32Z","abstract_excerpt":"Recently, Omni-modal large language models (OLLMs) have sparked a new wave of research, achieving impressive results in tasks such as audio-video understanding and real-time environment perception. However, hallucination issues still persist. Similar to the bimodal setting, the priors from the text modality tend to dominate, leading OLLMs to rely more heavily on textual cues while neglecting visual and audio information. In addition, fully multimodal scenarios introduce new challenges. Most existing models align visual or auditory modalities with text independently during training, while ignor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00723","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00723/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.00723","created_at":"2026-07-05T12:02:21.779413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.00723v1","created_at":"2026-07-05T12:02:21.779413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00723","created_at":"2026-07-05T12:02:21.779413+00:00"},{"alias_kind":"pith_short_12","alias_value":"JC6MAHAJI3Y5","created_at":"2026-07-05T12:02:21.779413+00:00"},{"alias_kind":"pith_short_16","alias_value":"JC6MAHAJI3Y5P5HZ","created_at":"2026-07-05T12:02:21.779413+00:00"},{"alias_kind":"pith_short_8","alias_value":"JC6MAHAJ","created_at":"2026-07-05T12:02:21.779413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.14129","citing_title":"Don't Let the Video Speak: Audio-Contrastive Preference Optimization for Audio-Visual Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14520","citing_title":"Chain of Modality: From Static Fusion to Dynamic Orchestration in Omni-MLLMs","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU","json":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU.json","graph_json":"https://pith.science/api/pith-number/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/graph.json","events_json":"https://pith.science/api/pith-number/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/events.json","paper":"https://pith.science/paper/JC6MAHAJ"},"agent_actions":{"view_html":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU","download_json":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU.json","view_paper":"https://pith.science/paper/JC6MAHAJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.00723&json=true","fetch_graph":"https://pith.science/api/pith-number/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/graph.json","fetch_events":"https://pith.science/api/pith-number/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/action/storage_attestation","attest_author":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/action/author_attestation","sign_citation":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/action/citation_signature","submit_replication":"https://pith.science/pith/JC6MAHAJI3Y5P5HZAVOM5Q2IIU/action/replication_record"}},"created_at":"2026-07-05T12:02:21.779413+00:00","updated_at":"2026-07-05T12:02:21.779413+00:00"}