{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:D4CQMV567ZI5DTUSC4GRE5P5UG","short_pith_number":"pith:D4CQMV56","schema_version":"1.0","canonical_sha256":"1f050657befe51d1ce92170d1275fda1813b201c46aece1cf1e9b8dbb4656fb9","source":{"kind":"arxiv","id":"2607.18718","version":1},"attestation_state":"computed","paper":{"title":"Summary of DCASE 2026 Task 5: Audio-Dependent Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Chunyat Wu, Haolin He, Jiahe Lei, Jian Liu, Jiayi Zhou, Mark D. Plumbley, Mingru Yang, Qiuqiang Kong, Renhe Sun, Runbang Wang, Weiqiang Wang, Xie Chen, Xingjian Du, Xiquan Li, Yun Chen, Zhengxi Liu, Zheqi Dai, Zhiyao Duan, Zining Liang","submitted_at":"2026-07-21T05:17:31Z","abstract_excerpt":"DCASE~2026 Task~5 introduces Audio-Dependent Question Answering (ADQA), which tests whether large audio-language models answer from the audio rather than from textual priors. An Audio-Dependency Filtering (ADF) pipeline combines silent-audio probing, per-option perplexity, a large language model (LLM) commonsense check, and human review to remove items solvable from text alone. The 3000 items that pass form the ADQA-Bench evaluation set, spanning music, speech, and environmental audio. The inaugural edition draws 14 teams and 36 submissions across two tracks defined by total parameter count (u"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.18718","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2026-07-21T05:17:31Z","cross_cats_sorted":[],"title_canon_sha256":"25579e33fcf322f989c841bf77c2bcd3d46195afd3ca0474eabd7f0d7d1a6e20","abstract_canon_sha256":"1f0c7a2394eadfc7ff4a550baa5c83b9acf4bcda53b5e81be3ab4e450f5380fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-22T01:23:08.242165Z","signature_b64":"NvnnbdHyLKSAdH5WpxWfXgG35EZ7qIHi8IOfqbr2Y/yES//nihInndlodNC9sT7Pozu5Ior63k6j8d8B5YufCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f050657befe51d1ce92170d1275fda1813b201c46aece1cf1e9b8dbb4656fb9","last_reissued_at":"2026-07-22T01:23:08.241298Z","signature_status":"signed_v1","first_computed_at":"2026-07-22T01:23:08.241298Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Summary of DCASE 2026 Task 5: Audio-Dependent Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Chunyat Wu, Haolin He, Jiahe Lei, Jian Liu, Jiayi Zhou, Mark D. Plumbley, Mingru Yang, Qiuqiang Kong, Renhe Sun, Runbang Wang, Weiqiang Wang, Xie Chen, Xingjian Du, Xiquan Li, Yun Chen, Zhengxi Liu, Zheqi Dai, Zhiyao Duan, Zining Liang","submitted_at":"2026-07-21T05:17:31Z","abstract_excerpt":"DCASE~2026 Task~5 introduces Audio-Dependent Question Answering (ADQA), which tests whether large audio-language models answer from the audio rather than from textual priors. An Audio-Dependency Filtering (ADF) pipeline combines silent-audio probing, per-option perplexity, a large language model (LLM) commonsense check, and human review to remove items solvable from text alone. The 3000 items that pass form the ADQA-Bench evaluation set, spanning music, speech, and environmental audio. The inaugural edition draws 14 teams and 36 submissions across two tracks defined by total parameter count (u"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.18718","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.18718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.18718","created_at":"2026-07-22T01:23:08.241758+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.18718v1","created_at":"2026-07-22T01:23:08.241758+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.18718","created_at":"2026-07-22T01:23:08.241758+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4CQMV567ZI5","created_at":"2026-07-22T01:23:08.241758+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4CQMV567ZI5DTUS","created_at":"2026-07-22T01:23:08.241758+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4CQMV56","created_at":"2026-07-22T01:23:08.241758+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG","json":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG.json","graph_json":"https://pith.science/api/pith-number/D4CQMV567ZI5DTUSC4GRE5P5UG/graph.json","events_json":"https://pith.science/api/pith-number/D4CQMV567ZI5DTUSC4GRE5P5UG/events.json","paper":"https://pith.science/paper/D4CQMV56"},"agent_actions":{"view_html":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG","download_json":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG.json","view_paper":"https://pith.science/paper/D4CQMV56","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.18718&json=true","fetch_graph":"https://pith.science/api/pith-number/D4CQMV567ZI5DTUSC4GRE5P5UG/graph.json","fetch_events":"https://pith.science/api/pith-number/D4CQMV567ZI5DTUSC4GRE5P5UG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG/action/storage_attestation","attest_author":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG/action/author_attestation","sign_citation":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG/action/citation_signature","submit_replication":"https://pith.science/pith/D4CQMV567ZI5DTUSC4GRE5P5UG/action/replication_record"}},"created_at":"2026-07-22T01:23:08.241758+00:00","updated_at":"2026-07-22T01:23:08.241758+00:00"}