{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NCOY2E5DBQOGU5ARRSRV6KJZJL","short_pith_number":"pith:NCOY2E5D","schema_version":"1.0","canonical_sha256":"689d8d13a30c1c6a74118ca35f29394ad2d2031a05352e3a01629b983d4d720c","source":{"kind":"arxiv","id":"2406.13275","version":2},"attestation_state":"computed","paper":{"title":"Enhancing Automated Audio Captioning via Large Language Models with Optimized Audio Encoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Bin Wang, Gang Li, Heinrich Dinkel, Jizhong Liu, Junbo Zhang, Yongqing Wang, Yujun Wang, Zhiyong Yan","submitted_at":"2024-06-19T07:09:46Z","abstract_excerpt":"Automated audio captioning (AAC) is an audio-to-text task to describe audio contents in natural language. Recently, the advancements in large language models (LLMs), with improvements in training approaches for audio encoders, have opened up possibilities for improving AAC. Thus, we explore enhancing AAC from three aspects: 1) a pre-trained audio encoder via consistent ensemble distillation (CED) is used to improve the effectivity of acoustic tokens, with a querying transformer (Q-Former) bridging the modality gap to LLM and compress acoustic tokens; 2) we investigate the advantages of using a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.13275","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-06-19T07:09:46Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"91c9bcdd5465ad8c460b0a5debf2cd95654e859688f37c352af99f4f5e81c8dc","abstract_canon_sha256":"e1a9f820ba7eda08d7d0874f9b406eb90f2f1cc12652ca62d4c3b2ccbee12f67"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:28.679971Z","signature_b64":"ErGxPLu2qm1mIsz9nwAADy+KFBIOiwQ2vv6xgnCLNnQhoJeEK+d4njlaztN25EDeh2HVPzlD+jJKeXHfZqlZAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"689d8d13a30c1c6a74118ca35f29394ad2d2031a05352e3a01629b983d4d720c","last_reissued_at":"2026-07-05T08:36:28.679513Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:28.679513Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Automated Audio Captioning via Large Language Models with Optimized Audio Encoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Bin Wang, Gang Li, Heinrich Dinkel, Jizhong Liu, Junbo Zhang, Yongqing Wang, Yujun Wang, Zhiyong Yan","submitted_at":"2024-06-19T07:09:46Z","abstract_excerpt":"Automated audio captioning (AAC) is an audio-to-text task to describe audio contents in natural language. Recently, the advancements in large language models (LLMs), with improvements in training approaches for audio encoders, have opened up possibilities for improving AAC. Thus, we explore enhancing AAC from three aspects: 1) a pre-trained audio encoder via consistent ensemble distillation (CED) is used to improve the effectivity of acoustic tokens, with a querying transformer (Q-Former) bridging the modality gap to LLM and compress acoustic tokens; 2) we investigate the advantages of using a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.13275","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.13275/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.13275","created_at":"2026-07-05T08:36:28.679571+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.13275v2","created_at":"2026-07-05T08:36:28.679571+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.13275","created_at":"2026-07-05T08:36:28.679571+00:00"},{"alias_kind":"pith_short_12","alias_value":"NCOY2E5DBQOG","created_at":"2026-07-05T08:36:28.679571+00:00"},{"alias_kind":"pith_short_16","alias_value":"NCOY2E5DBQOGU5AR","created_at":"2026-07-05T08:36:28.679571+00:00"},{"alias_kind":"pith_short_8","alias_value":"NCOY2E5D","created_at":"2026-07-05T08:36:28.679571+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL","json":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL.json","graph_json":"https://pith.science/api/pith-number/NCOY2E5DBQOGU5ARRSRV6KJZJL/graph.json","events_json":"https://pith.science/api/pith-number/NCOY2E5DBQOGU5ARRSRV6KJZJL/events.json","paper":"https://pith.science/paper/NCOY2E5D"},"agent_actions":{"view_html":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL","download_json":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL.json","view_paper":"https://pith.science/paper/NCOY2E5D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.13275&json=true","fetch_graph":"https://pith.science/api/pith-number/NCOY2E5DBQOGU5ARRSRV6KJZJL/graph.json","fetch_events":"https://pith.science/api/pith-number/NCOY2E5DBQOGU5ARRSRV6KJZJL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL/action/storage_attestation","attest_author":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL/action/author_attestation","sign_citation":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL/action/citation_signature","submit_replication":"https://pith.science/pith/NCOY2E5DBQOGU5ARRSRV6KJZJL/action/replication_record"}},"created_at":"2026-07-05T08:36:28.679571+00:00","updated_at":"2026-07-05T08:36:28.679571+00:00"}