{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SQH7GIPV7O4CKGZO723SCKK3MS","short_pith_number":"pith:SQH7GIPV","schema_version":"1.0","canonical_sha256":"940ff321f5fbb8251b2efeb721295b64b40513ab517ee8f8fd0ce208ef96cace","source":{"kind":"arxiv","id":"2210.08624","version":2},"attestation_state":"computed","paper":{"title":"Attention-Based Audio Embeddings for Query-by-Example","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Anup Singh, Kris Demuynck, Vipul Arora","submitted_at":"2022-10-16T19:37:55Z","abstract_excerpt":"An ideal audio retrieval system efficiently and robustly recognizes a short query snippet from an extensive database. However, the performance of well-known audio fingerprinting systems falls short at high signal distortion levels. This paper presents an audio retrieval system that generates noise and reverberation robust audio fingerprints using the contrastive learning framework. Using these fingerprints, the method performs a comprehensive search to identify the query audio and precisely estimate its timestamp in the reference audio. Our framework involves training a CNN to maximize the sim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.08624","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2022-10-16T19:37:55Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"2745374d2f5c01873cafba7c51569b27b01981d22a4bd55853f1c3b769890b1d","abstract_canon_sha256":"70f90b398d1cbf48373adcfe60aa2865ee2760b293c3d7e029fdae6b091930f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:20.067483Z","signature_b64":"ejEwMuQXRAv2rq2mGduA2JMSZrLh1htSv1kHnRLchk6bnDZ13AbfS4X4WjhrHOeNnr3KLJgieW473vMtrkx+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"940ff321f5fbb8251b2efeb721295b64b40513ab517ee8f8fd0ce208ef96cace","last_reissued_at":"2026-07-05T09:38:20.066988Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:20.066988Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attention-Based Audio Embeddings for Query-by-Example","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Anup Singh, Kris Demuynck, Vipul Arora","submitted_at":"2022-10-16T19:37:55Z","abstract_excerpt":"An ideal audio retrieval system efficiently and robustly recognizes a short query snippet from an extensive database. However, the performance of well-known audio fingerprinting systems falls short at high signal distortion levels. This paper presents an audio retrieval system that generates noise and reverberation robust audio fingerprints using the contrastive learning framework. Using these fingerprints, the method performs a comprehensive search to identify the query audio and precisely estimate its timestamp in the reference audio. Our framework involves training a CNN to maximize the sim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.08624","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.08624/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.08624","created_at":"2026-07-05T09:38:20.067050+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.08624v2","created_at":"2026-07-05T09:38:20.067050+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.08624","created_at":"2026-07-05T09:38:20.067050+00:00"},{"alias_kind":"pith_short_12","alias_value":"SQH7GIPV7O4C","created_at":"2026-07-05T09:38:20.067050+00:00"},{"alias_kind":"pith_short_16","alias_value":"SQH7GIPV7O4CKGZO","created_at":"2026-07-05T09:38:20.067050+00:00"},{"alias_kind":"pith_short_8","alias_value":"SQH7GIPV","created_at":"2026-07-05T09:38:20.067050+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26824","citing_title":"wav2tok 2.0: Scalable Audio Tokenization Maintaining Explicit Pairwise Token Alignment for Efficient Audio Retrieval","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06582","citing_title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06582","citing_title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS","json":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS.json","graph_json":"https://pith.science/api/pith-number/SQH7GIPV7O4CKGZO723SCKK3MS/graph.json","events_json":"https://pith.science/api/pith-number/SQH7GIPV7O4CKGZO723SCKK3MS/events.json","paper":"https://pith.science/paper/SQH7GIPV"},"agent_actions":{"view_html":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS","download_json":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS.json","view_paper":"https://pith.science/paper/SQH7GIPV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.08624&json=true","fetch_graph":"https://pith.science/api/pith-number/SQH7GIPV7O4CKGZO723SCKK3MS/graph.json","fetch_events":"https://pith.science/api/pith-number/SQH7GIPV7O4CKGZO723SCKK3MS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS/action/storage_attestation","attest_author":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS/action/author_attestation","sign_citation":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS/action/citation_signature","submit_replication":"https://pith.science/pith/SQH7GIPV7O4CKGZO723SCKK3MS/action/replication_record"}},"created_at":"2026-07-05T09:38:20.067050+00:00","updated_at":"2026-07-05T09:38:20.067050+00:00"}