{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SMJAPFB5C6JW5A5LPJVHYELPFL","short_pith_number":"pith:SMJAPFB5","schema_version":"1.0","canonical_sha256":"931207943d17936e83ab7a6a7c116f2af75da09ed48bf431b3212c07705b75a3","source":{"kind":"arxiv","id":"2209.14275","version":1},"attestation_state":"computed","paper":{"title":"Audio Retrieval with WavText5K and CLAP Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"eess.AS","authors_text":"Benjamin Elizalde, Huaming Wang, Soham Deshmukh","submitted_at":"2022-09-28T17:39:26Z","abstract_excerpt":"Audio-Text retrieval takes a natural language query to retrieve relevant audio files in a database. Conversely, Text-Audio retrieval takes an audio file as a query to retrieve relevant natural language descriptions. Most of the literature train retrieval systems with one audio captioning dataset, but evaluating the benefit of training with multiple datasets is underexplored. Moreover, retrieval systems have to learn the alignment between elaborated sentences describing audio content of variable length ranging from a few seconds to several minutes. In this work, we propose a new collection of w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.14275","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2022-09-28T17:39:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"689f1eba60cd8e232e6f3aea93e2f905c8141ad0a520e2fca273e9cb9b7148a0","abstract_canon_sha256":"c43301e71d3b85324feaf081e2e8fadbadf00f16831b88f5e6c72e280d452584"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:01:49.444723Z","signature_b64":"16WbZB7fRzOzm3lQH5Q8oZFa5TLgJRZn4k3zUBrrk59FlhA3RtxSpQrEKvEdXWi+PFzB2DB41zLt0knQT7BeAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"931207943d17936e83ab7a6a7c116f2af75da09ed48bf431b3212c07705b75a3","last_reissued_at":"2026-07-05T05:01:49.444387Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:01:49.444387Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio Retrieval with WavText5K and CLAP Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"eess.AS","authors_text":"Benjamin Elizalde, Huaming Wang, Soham Deshmukh","submitted_at":"2022-09-28T17:39:26Z","abstract_excerpt":"Audio-Text retrieval takes a natural language query to retrieve relevant audio files in a database. Conversely, Text-Audio retrieval takes an audio file as a query to retrieve relevant natural language descriptions. Most of the literature train retrieval systems with one audio captioning dataset, but evaluating the benefit of training with multiple datasets is underexplored. Moreover, retrieval systems have to learn the alignment between elaborated sentences describing audio content of variable length ranging from a few seconds to several minutes. In this work, we propose a new collection of w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.14275","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.14275/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.14275","created_at":"2026-07-05T05:01:49.444442+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.14275v1","created_at":"2026-07-05T05:01:49.444442+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.14275","created_at":"2026-07-05T05:01:49.444442+00:00"},{"alias_kind":"pith_short_12","alias_value":"SMJAPFB5C6JW","created_at":"2026-07-05T05:01:49.444442+00:00"},{"alias_kind":"pith_short_16","alias_value":"SMJAPFB5C6JW5A5L","created_at":"2026-07-05T05:01:49.444442+00:00"},{"alias_kind":"pith_short_8","alias_value":"SMJAPFB5","created_at":"2026-07-05T05:01:49.444442+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.08128","citing_title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03361","citing_title":"ReasonAudio: A Benchmark for Evaluating Reasoning Beyond Matching in Text-Audio Retrieval","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00329","citing_title":"Fast Text-to-Audio Generation with One-Step Sampling via Energy-Scoring and Auxiliary Contextual Representation Distillation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03361","citing_title":"ReasonAudio: A Benchmark for Evaluating Reasoning Beyond Matching in Text-Audio Retrieval","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL","json":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL.json","graph_json":"https://pith.science/api/pith-number/SMJAPFB5C6JW5A5LPJVHYELPFL/graph.json","events_json":"https://pith.science/api/pith-number/SMJAPFB5C6JW5A5LPJVHYELPFL/events.json","paper":"https://pith.science/paper/SMJAPFB5"},"agent_actions":{"view_html":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL","download_json":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL.json","view_paper":"https://pith.science/paper/SMJAPFB5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.14275&json=true","fetch_graph":"https://pith.science/api/pith-number/SMJAPFB5C6JW5A5LPJVHYELPFL/graph.json","fetch_events":"https://pith.science/api/pith-number/SMJAPFB5C6JW5A5LPJVHYELPFL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL/action/storage_attestation","attest_author":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL/action/author_attestation","sign_citation":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL/action/citation_signature","submit_replication":"https://pith.science/pith/SMJAPFB5C6JW5A5LPJVHYELPFL/action/replication_record"}},"created_at":"2026-07-05T05:01:49.444442+00:00","updated_at":"2026-07-05T05:01:49.444442+00:00"}