{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZRIENHZE6F4UH6RJNMYPYNTGUA","short_pith_number":"pith:ZRIENHZE","schema_version":"1.0","canonical_sha256":"cc50469f24f17943fa296b30fc3666a00e4516898b35ec6b8238a81848102c6a","source":{"kind":"arxiv","id":"2506.07233","version":2},"attestation_state":"computed","paper":{"title":"Reducing Object Hallucination in Large Audio-Language Models via Audio-Aware Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Cheng-Han Chiang, Hung-yi Lee, Ke-Han Lu, Tzu-Wen Hsu","submitted_at":"2025-06-08T17:36:50Z","abstract_excerpt":"Large Audio-Language Models (LALMs) can take audio and text as the inputs and answer questions about the audio. While prior LALMs have shown strong performance on standard benchmarks, there has been alarming evidence that LALMs can hallucinate what is presented in the audio. To mitigate the hallucination of LALMs, we introduce Audio-Aware Decoding (AAD), a lightweight inference-time strategy that uses contrastive decoding to compare the token prediction logits with and without the audio context. By contrastive decoding, AAD promotes the tokens whose probability increases when the audio is pres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07233","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2025-06-08T17:36:50Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"24aacf5bdae78fb77b952cb666bc524978945c40c26cbfd5c75a44558c11d3ef","abstract_canon_sha256":"c4e363e7cece37326b7f1859c36cf32f1ee07cf9122ce8a7f7d6dd83b8071fb9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:18.582860Z","signature_b64":"BVDA1WRmaiLDXYKfJrv6WGOwi+8kl5Xrf5Ca3qHyqy1k+ixGfJjA0Z6xf2RREBHDHHL7+eLv/py2SjntCIJ9Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc50469f24f17943fa296b30fc3666a00e4516898b35ec6b8238a81848102c6a","last_reissued_at":"2026-07-05T12:11:18.582294Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:18.582294Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reducing Object Hallucination in Large Audio-Language Models via Audio-Aware Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Cheng-Han Chiang, Hung-yi Lee, Ke-Han Lu, Tzu-Wen Hsu","submitted_at":"2025-06-08T17:36:50Z","abstract_excerpt":"Large Audio-Language Models (LALMs) can take audio and text as the inputs and answer questions about the audio. While prior LALMs have shown strong performance on standard benchmarks, there has been alarming evidence that LALMs can hallucinate what is presented in the audio. To mitigate the hallucination of LALMs, we introduce Audio-Aware Decoding (AAD), a lightweight inference-time strategy that uses contrastive decoding to compare the token prediction logits with and without the audio context. By contrastive decoding, AAD promotes the tokens whose probability increases when the audio is pres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07233","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07233/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07233","created_at":"2026-07-05T12:11:18.582366+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07233v2","created_at":"2026-07-05T12:11:18.582366+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07233","created_at":"2026-07-05T12:11:18.582366+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZRIENHZE6F4U","created_at":"2026-07-05T12:11:18.582366+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZRIENHZE6F4UH6RJ","created_at":"2026-07-05T12:11:18.582366+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZRIENHZE","created_at":"2026-07-05T12:11:18.582366+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23052","citing_title":"CAAD: Contrastive Audio-Aware Distillation for Efficient Speech Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13441","citing_title":"Why Sampling Is Not Choosing: Intentionality, Agency, and Moral Responsibility in Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00247","citing_title":"Adaptive Perturbation Selection for Contrastive Audio Decoding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01766","citing_title":"Mitigating Multimodal LLMs Hallucinations via Relevance Propagation at Inference Time","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10065","citing_title":"ASPIRin: Action Space Projection for Interactivity-Optimized Reinforcement Learning in Full-Duplex Speech Language Models","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA","json":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA.json","graph_json":"https://pith.science/api/pith-number/ZRIENHZE6F4UH6RJNMYPYNTGUA/graph.json","events_json":"https://pith.science/api/pith-number/ZRIENHZE6F4UH6RJNMYPYNTGUA/events.json","paper":"https://pith.science/paper/ZRIENHZE"},"agent_actions":{"view_html":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA","download_json":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA.json","view_paper":"https://pith.science/paper/ZRIENHZE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07233&json=true","fetch_graph":"https://pith.science/api/pith-number/ZRIENHZE6F4UH6RJNMYPYNTGUA/graph.json","fetch_events":"https://pith.science/api/pith-number/ZRIENHZE6F4UH6RJNMYPYNTGUA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA/action/storage_attestation","attest_author":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA/action/author_attestation","sign_citation":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA/action/citation_signature","submit_replication":"https://pith.science/pith/ZRIENHZE6F4UH6RJNMYPYNTGUA/action/replication_record"}},"created_at":"2026-07-05T12:11:18.582366+00:00","updated_at":"2026-07-05T12:11:18.582366+00:00"}