{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZOOEOT3YCU7552P4IYTU52QPIK","short_pith_number":"pith:ZOOEOT3Y","schema_version":"1.0","canonical_sha256":"cb9c474f78153fdee9fc46274eea0f42a712cf3d95c9764b23597b870261e7c7","source":{"kind":"arxiv","id":"2501.07246","version":1},"attestation_state":"computed","paper":{"title":"Audio-CoT: Exploring Chain-of-Thought Reasoning in Large Audio Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Eng Siong Chng, Xie Chen, Yuping Wang, Zhuo Chen, Ziyang Ma","submitted_at":"2025-01-13T11:54:40Z","abstract_excerpt":"Large Audio-Language Models (LALMs) have demonstrated remarkable performance in tasks involving audio perception and understanding, such as speech recognition and audio captioning. However, their reasoning capabilities - critical for solving complex real-world problems - remain underexplored. In this work, we conduct the first exploration into integrating Chain-of-Thought (CoT) reasoning into LALMs to enhance their reasoning ability across auditory modalities. We evaluate representative CoT methods, analyzing their performance in both information extraction and reasoning tasks across sound, mu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.07246","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2025-01-13T11:54:40Z","cross_cats_sorted":["cs.CL","cs.MM","eess.AS"],"title_canon_sha256":"8efbc4a4abe78ae3ddb5c52b437fe30c018844ec1de7275a62b87f0af00637a0","abstract_canon_sha256":"4e4c4e2aeedea12f5a1782f00a08d3eb80139c93d88a89f6f499cb0a9ba0d80f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:17.872797Z","signature_b64":"smK7zWnLy2hjekxUFZ3N575bf48vAGrxM01wxKgfWZOzGS2zRJCHqBZh68jhxgQofPxgyBtSONp51lvcPWF1Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb9c474f78153fdee9fc46274eea0f42a712cf3d95c9764b23597b870261e7c7","last_reissued_at":"2026-07-05T10:00:17.872371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:17.872371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio-CoT: Exploring Chain-of-Thought Reasoning in Large Audio Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Eng Siong Chng, Xie Chen, Yuping Wang, Zhuo Chen, Ziyang Ma","submitted_at":"2025-01-13T11:54:40Z","abstract_excerpt":"Large Audio-Language Models (LALMs) have demonstrated remarkable performance in tasks involving audio perception and understanding, such as speech recognition and audio captioning. However, their reasoning capabilities - critical for solving complex real-world problems - remain underexplored. In this work, we conduct the first exploration into integrating Chain-of-Thought (CoT) reasoning into LALMs to enhance their reasoning ability across auditory modalities. We evaluate representative CoT methods, analyzing their performance in both information extraction and reasoning tasks across sound, mu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.07246","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.07246/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.07246","created_at":"2026-07-05T10:00:17.872430+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.07246v1","created_at":"2026-07-05T10:00:17.872430+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.07246","created_at":"2026-07-05T10:00:17.872430+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZOOEOT3YCU75","created_at":"2026-07-05T10:00:17.872430+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZOOEOT3YCU7552P4","created_at":"2026-07-05T10:00:17.872430+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZOOEOT3Y","created_at":"2026-07-05T10:00:17.872430+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10838","citing_title":"Towards Deep Contextual Reasoning from Broad Descriptions for ASR with Speech-LLM via Metadata-Driven Reasoning Chains","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08425","citing_title":"TinyGiantALM: A Compact Audio-Language Model for Intent-Aware Reasoning under Resource Constraints","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07264","citing_title":"VISA: A Visual Information Strengthened Audio-Reasoning System for the Interspeech 2026 ARC Agent Track","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27190","citing_title":"Learning When to Think While Listening in Large Audio-Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21008","citing_title":"A Survey of Audio Reasoning in Multimodal Foundation Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2507.08128","citing_title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25591","citing_title":"Walking Through Uncertainty: An Empirical Study of Uncertainty Estimation for Audio-Aware Large Language Models","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04613","citing_title":"VocalParse: Towards Unified and Scalable Singing Voice Transcription with Large Audio Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08363","citing_title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12527","citing_title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK","json":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK.json","graph_json":"https://pith.science/api/pith-number/ZOOEOT3YCU7552P4IYTU52QPIK/graph.json","events_json":"https://pith.science/api/pith-number/ZOOEOT3YCU7552P4IYTU52QPIK/events.json","paper":"https://pith.science/paper/ZOOEOT3Y"},"agent_actions":{"view_html":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK","download_json":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK.json","view_paper":"https://pith.science/paper/ZOOEOT3Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.07246&json=true","fetch_graph":"https://pith.science/api/pith-number/ZOOEOT3YCU7552P4IYTU52QPIK/graph.json","fetch_events":"https://pith.science/api/pith-number/ZOOEOT3YCU7552P4IYTU52QPIK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK/action/storage_attestation","attest_author":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK/action/author_attestation","sign_citation":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK/action/citation_signature","submit_replication":"https://pith.science/pith/ZOOEOT3YCU7552P4IYTU52QPIK/action/replication_record"}},"created_at":"2026-07-05T10:00:17.872430+00:00","updated_at":"2026-07-05T10:00:17.872430+00:00"}