{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RWKJU5PLGPU45LJXEAET24YKET","short_pith_number":"pith:RWKJU5PL","schema_version":"1.0","canonical_sha256":"8d949a75eb33e9cead3720093d730a24f0b85d95d2d56269e991f0a4095f7e29","source":{"kind":"arxiv","id":"2305.10790","version":3},"attestation_state":"computed","paper":{"title":"Listen, Think, and Understand","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Alexander H. Liu, Hongyin Luo, James Glass, Leonid Karlinsky, Yuan Gong","submitted_at":"2023-05-18T08:03:37Z","abstract_excerpt":"The ability of artificial intelligence (AI) systems to perceive and comprehend audio signals is crucial for many applications. Although significant progress has been made in this area since the development of AudioSet, most existing models are designed to map audio inputs to pre-defined, discrete sound label sets. In contrast, humans possess the ability to not only classify sounds into general categories, but also to listen to the finer details of the sounds, explain the reason for the predictions, think about what the sound infers, and understand the scene and what action needs to be taken, i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.10790","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2023-05-18T08:03:37Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"eb9fb9879a71e0a909575acc29b7ddba9b51ae2a9fb39d8bd4fbf05a689e3a7b","abstract_canon_sha256":"0bf82dd48ce51b90b21a78e25d2521c1c6b28fa8d1d633df4970239f912bda53"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:47:02.496396Z","signature_b64":"Lvrykr2QG0T5IWBqvN9R/IV/fgidVUXBwf61adTcMi55T0PvzSYNGY+ioebK+tpx9japewufnnRvKaykoNe0Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d949a75eb33e9cead3720093d730a24f0b85d95d2d56269e991f0a4095f7e29","last_reissued_at":"2026-07-05T07:47:02.495938Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:47:02.495938Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Listen, Think, and Understand","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Alexander H. Liu, Hongyin Luo, James Glass, Leonid Karlinsky, Yuan Gong","submitted_at":"2023-05-18T08:03:37Z","abstract_excerpt":"The ability of artificial intelligence (AI) systems to perceive and comprehend audio signals is crucial for many applications. Although significant progress has been made in this area since the development of AudioSet, most existing models are designed to map audio inputs to pre-defined, discrete sound label sets. In contrast, humans possess the ability to not only classify sounds into general categories, but also to listen to the finer details of the sounds, explain the reason for the predictions, think about what the sound infers, and understand the scene and what action needs to be taken, i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.10790","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.10790/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.10790","created_at":"2026-07-05T07:47:02.495994+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.10790v3","created_at":"2026-07-05T07:47:02.495994+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.10790","created_at":"2026-07-05T07:47:02.495994+00:00"},{"alias_kind":"pith_short_12","alias_value":"RWKJU5PLGPU4","created_at":"2026-07-05T07:47:02.495994+00:00"},{"alias_kind":"pith_short_16","alias_value":"RWKJU5PLGPU45LJX","created_at":"2026-07-05T07:47:02.495994+00:00"},{"alias_kind":"pith_short_8","alias_value":"RWKJU5PL","created_at":"2026-07-05T07:47:02.495994+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":24,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25391","citing_title":"From Sounds to Scenes: A Benchmark for Evaluating Context-Aware Auditory Scene Understanding in Large Audio Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10147","citing_title":"From Senses to Decisions: The Information Flow of Auditory and Visual Perception in Multimodal LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18273","citing_title":"Continuous Audio Thinking for Large Audio Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00247","citing_title":"Adaptive Perturbation Selection for Contrastive Audio Decoding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20519","citing_title":"Codec-Robust Attacks on Audio LLMs","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20519","citing_title":"Codec-Robust Attacks on Audio LLMs","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19101","citing_title":"Heterogeneity-Aware Dataset Scheduling for Efficient Audio Large Language Model Training","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2310.13289","citing_title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2507.16632","citing_title":"Step-Audio 2 Technical Report","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12527","citing_title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET","json":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET.json","graph_json":"https://pith.science/api/pith-number/RWKJU5PLGPU45LJXEAET24YKET/graph.json","events_json":"https://pith.science/api/pith-number/RWKJU5PLGPU45LJXEAET24YKET/events.json","paper":"https://pith.science/paper/RWKJU5PL"},"agent_actions":{"view_html":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET","download_json":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET.json","view_paper":"https://pith.science/paper/RWKJU5PL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.10790&json=true","fetch_graph":"https://pith.science/api/pith-number/RWKJU5PLGPU45LJXEAET24YKET/graph.json","fetch_events":"https://pith.science/api/pith-number/RWKJU5PLGPU45LJXEAET24YKET/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET/action/storage_attestation","attest_author":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET/action/author_attestation","sign_citation":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET/action/citation_signature","submit_replication":"https://pith.science/pith/RWKJU5PLGPU45LJXEAET24YKET/action/replication_record"}},"created_at":"2026-07-05T07:47:02.495994+00:00","updated_at":"2026-07-05T07:47:02.495994+00:00"}