{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XD5WD22CFIMH4M6EBGBCWHZYGI","short_pith_number":"pith:XD5WD22C","schema_version":"1.0","canonical_sha256":"b8fb61eb422a187e33c409822b1f3832182182698af9c378546350a15afb2bb8","source":{"kind":"arxiv","id":"2406.16990","version":2},"attestation_state":"computed","paper":{"title":"AND: Audio Network Dissection for Interpreting Deep Acoustic Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Tsui-Wei Weng, Tung-yu Wu, Yu-Xiang Lin","submitted_at":"2024-06-24T06:02:07Z","abstract_excerpt":"Neuron-level interpretations aim to explain network behaviors and properties by investigating neurons responsive to specific perceptual or structural input patterns. Although there is emerging work in the vision and language domains, none is explored for acoustic models. To bridge the gap, we introduce $\\textit{AND}$, the first $\\textbf{A}$udio $\\textbf{N}$etwork $\\textbf{D}$issection framework that automatically establishes natural language explanations of acoustic neurons based on highly-responsive audio. $\\textit{AND}$ features the use of LLMs to summarize mutual acoustic features and ident"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16990","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-06-24T06:02:07Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"0b8bf23d4b8b304cbfeab6f36e782bde9278d524f1d6060df8c11bb75f4953f8","abstract_canon_sha256":"10cb9cee6d3d82bfd0f3da40283790fa5493b4e314318757ba70d4d18e73effe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:37.479067Z","signature_b64":"hQBdcuAZAt2GakMqdShpwbPIsesEqaXX5C+szwvcP+/jFSlq+Jeilc8jelFd5X2cVTCIwROGj671sla3rMmYAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8fb61eb422a187e33c409822b1f3832182182698af9c378546350a15afb2bb8","last_reissued_at":"2026-07-05T08:42:37.478587Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:37.478587Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AND: Audio Network Dissection for Interpreting Deep Acoustic Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Tsui-Wei Weng, Tung-yu Wu, Yu-Xiang Lin","submitted_at":"2024-06-24T06:02:07Z","abstract_excerpt":"Neuron-level interpretations aim to explain network behaviors and properties by investigating neurons responsive to specific perceptual or structural input patterns. Although there is emerging work in the vision and language domains, none is explored for acoustic models. To bridge the gap, we introduce $\\textit{AND}$, the first $\\textbf{A}$udio $\\textbf{N}$etwork $\\textbf{D}$issection framework that automatically establishes natural language explanations of acoustic neurons based on highly-responsive audio. $\\textit{AND}$ features the use of LLMs to summarize mutual acoustic features and ident"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16990","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16990/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16990","created_at":"2026-07-05T08:42:37.478646+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16990v2","created_at":"2026-07-05T08:42:37.478646+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16990","created_at":"2026-07-05T08:42:37.478646+00:00"},{"alias_kind":"pith_short_12","alias_value":"XD5WD22CFIMH","created_at":"2026-07-05T08:42:37.478646+00:00"},{"alias_kind":"pith_short_16","alias_value":"XD5WD22CFIMH4M6E","created_at":"2026-07-05T08:42:37.478646+00:00"},{"alias_kind":"pith_short_8","alias_value":"XD5WD22C","created_at":"2026-07-05T08:42:37.478646+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07907","citing_title":"Multimodal Unlearning Across Vision, Language, Video, and Audio: Survey of Methods, Datasets, and Benchmarks","ref_index":55,"is_internal_anchor":true},{"citing_arxiv_id":"2602.24176","citing_title":"Beyond Explainable AI (XAI): An Overdue Paradigm Shift and Post-XAI Research Directions","ref_index":263,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI","json":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI.json","graph_json":"https://pith.science/api/pith-number/XD5WD22CFIMH4M6EBGBCWHZYGI/graph.json","events_json":"https://pith.science/api/pith-number/XD5WD22CFIMH4M6EBGBCWHZYGI/events.json","paper":"https://pith.science/paper/XD5WD22C"},"agent_actions":{"view_html":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI","download_json":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI.json","view_paper":"https://pith.science/paper/XD5WD22C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16990&json=true","fetch_graph":"https://pith.science/api/pith-number/XD5WD22CFIMH4M6EBGBCWHZYGI/graph.json","fetch_events":"https://pith.science/api/pith-number/XD5WD22CFIMH4M6EBGBCWHZYGI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI/action/storage_attestation","attest_author":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI/action/author_attestation","sign_citation":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI/action/citation_signature","submit_replication":"https://pith.science/pith/XD5WD22CFIMH4M6EBGBCWHZYGI/action/replication_record"}},"created_at":"2026-07-05T08:42:37.478646+00:00","updated_at":"2026-07-05T08:42:37.478646+00:00"}