{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IIMUPYP4RZRVMP443QCX6LWMB5","short_pith_number":"pith:IIMUPYP4","schema_version":"1.0","canonical_sha256":"421947e1fc8e63563f9cdc057f2ecc0f5193e567e52735a07e2fad4057decbe3","source":{"kind":"arxiv","id":"2309.07372","version":1},"attestation_state":"computed","paper":{"title":"Training Audio Captioning Models without Audio","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Benjamin Elizalde, Bhiksha Raj, Dimitra Emmanouilidou, Huaming Wang, Rita Singh, Soham Deshmukh","submitted_at":"2023-09-14T01:16:02Z","abstract_excerpt":"Automated Audio Captioning (AAC) is the task of generating natural language descriptions given an audio stream. A typical AAC system requires manually curated training data of audio segments and corresponding text caption annotations. The creation of these audio-caption pairs is costly, resulting in general data scarcity for the task. In this work, we address this major limitation and propose an approach to train AAC systems using only text. Our approach leverages the multimodal space of contrastively trained audio-text models, such as CLAP. During training, a decoder generates captions condit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.07372","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2023-09-14T01:16:02Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"a5c99b0f3cbaced4fbc9fc722c75253b2efb0f5761aa32769a5b3c25067b928f","abstract_canon_sha256":"f7766018c0b2f47fceb3a2e96f71f518be55fc470c6e24e81b6397b5cecce26a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:50:34.529382Z","signature_b64":"kzNqj86p3l4Kti7OOy9vrP11Bf19WTSNWvraQ8heAihdYYi2swDHvQIWAh0Sfr0JfNp9Qe/GkE+Pw/t3Tk0bAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"421947e1fc8e63563f9cdc057f2ecc0f5193e567e52735a07e2fad4057decbe3","last_reissued_at":"2026-07-05T06:50:34.528890Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:50:34.528890Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Audio Captioning Models without Audio","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Benjamin Elizalde, Bhiksha Raj, Dimitra Emmanouilidou, Huaming Wang, Rita Singh, Soham Deshmukh","submitted_at":"2023-09-14T01:16:02Z","abstract_excerpt":"Automated Audio Captioning (AAC) is the task of generating natural language descriptions given an audio stream. A typical AAC system requires manually curated training data of audio segments and corresponding text caption annotations. The creation of these audio-caption pairs is costly, resulting in general data scarcity for the task. In this work, we address this major limitation and propose an approach to train AAC systems using only text. Our approach leverages the multimodal space of contrastively trained audio-text models, such as CLAP. During training, a decoder generates captions condit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.07372","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.07372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.07372","created_at":"2026-07-05T06:50:34.528953+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.07372v1","created_at":"2026-07-05T06:50:34.528953+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.07372","created_at":"2026-07-05T06:50:34.528953+00:00"},{"alias_kind":"pith_short_12","alias_value":"IIMUPYP4RZRV","created_at":"2026-07-05T06:50:34.528953+00:00"},{"alias_kind":"pith_short_16","alias_value":"IIMUPYP4RZRVMP44","created_at":"2026-07-05T06:50:34.528953+00:00"},{"alias_kind":"pith_short_8","alias_value":"IIMUPYP4","created_at":"2026-07-05T06:50:34.528953+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":240,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5","json":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5.json","graph_json":"https://pith.science/api/pith-number/IIMUPYP4RZRVMP443QCX6LWMB5/graph.json","events_json":"https://pith.science/api/pith-number/IIMUPYP4RZRVMP443QCX6LWMB5/events.json","paper":"https://pith.science/paper/IIMUPYP4"},"agent_actions":{"view_html":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5","download_json":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5.json","view_paper":"https://pith.science/paper/IIMUPYP4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.07372&json=true","fetch_graph":"https://pith.science/api/pith-number/IIMUPYP4RZRVMP443QCX6LWMB5/graph.json","fetch_events":"https://pith.science/api/pith-number/IIMUPYP4RZRVMP443QCX6LWMB5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5/action/storage_attestation","attest_author":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5/action/author_attestation","sign_citation":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5/action/citation_signature","submit_replication":"https://pith.science/pith/IIMUPYP4RZRVMP443QCX6LWMB5/action/replication_record"}},"created_at":"2026-07-05T06:50:34.528953+00:00","updated_at":"2026-07-05T06:50:34.528953+00:00"}