{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JYZHLVJFOYKX2DPW4IBMZKM7OX","short_pith_number":"pith:JYZHLVJF","schema_version":"1.0","canonical_sha256":"4e3275d52576157d0df6e202cca99f75c37f90c0f26fc2b5396eca0f94ed07e1","source":{"kind":"arxiv","id":"2207.03927","version":2},"attestation_state":"computed","paper":{"title":"BAST: Binaural Audio Spectrogram Transformer for Binaural Sound Localization","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jie Shi, Kiki van der Heijden, Sheng Kuang, Siamak Mehrkanoon","submitted_at":"2022-07-08T14:27:52Z","abstract_excerpt":"Accurate sound localization in a reverberation environment is essential for human auditory perception. Recently, Convolutional Neural Networks (CNNs) have been utilized to model the binaural human auditory pathway. However, CNN shows barriers in capturing the global acoustic features. To address this issue, we propose a novel end-to-end Binaural Audio Spectrogram Transformer (BAST) model to predict the sound azimuth in both anechoic and reverberation environments. Two modes of implementation, i.e. BAST-SP and BAST-NSP corresponding to BAST model with shared and non-shared parameters respective"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.03927","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2022-07-08T14:27:52Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"129039fbdf936523a4a2f470bdb9069431de0869f637260af365d4bd9fc5432b","abstract_canon_sha256":"f28df5c32ad1495dbef09183691f975d07822b7ac3b1333aa87406365ed67645"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:53:07.146900Z","signature_b64":"b5yz4gxjCMAEdWMKF5oLb4DTnfhaJYKiIUEsg+EtoLGuwfsG047+oHd6w7bLMv5lc1w/VOFrJTVF0x8KdcXzBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e3275d52576157d0df6e202cca99f75c37f90c0f26fc2b5396eca0f94ed07e1","last_reissued_at":"2026-07-05T08:53:07.146446Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:53:07.146446Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BAST: Binaural Audio Spectrogram Transformer for Binaural Sound Localization","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jie Shi, Kiki van der Heijden, Sheng Kuang, Siamak Mehrkanoon","submitted_at":"2022-07-08T14:27:52Z","abstract_excerpt":"Accurate sound localization in a reverberation environment is essential for human auditory perception. Recently, Convolutional Neural Networks (CNNs) have been utilized to model the binaural human auditory pathway. However, CNN shows barriers in capturing the global acoustic features. To address this issue, we propose a novel end-to-end Binaural Audio Spectrogram Transformer (BAST) model to predict the sound azimuth in both anechoic and reverberation environments. Two modes of implementation, i.e. BAST-SP and BAST-NSP corresponding to BAST model with shared and non-shared parameters respective"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.03927","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.03927/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.03927","created_at":"2026-07-05T08:53:07.146508+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.03927v2","created_at":"2026-07-05T08:53:07.146508+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.03927","created_at":"2026-07-05T08:53:07.146508+00:00"},{"alias_kind":"pith_short_12","alias_value":"JYZHLVJFOYKX","created_at":"2026-07-05T08:53:07.146508+00:00"},{"alias_kind":"pith_short_16","alias_value":"JYZHLVJFOYKX2DPW","created_at":"2026-07-05T08:53:07.146508+00:00"},{"alias_kind":"pith_short_8","alias_value":"JYZHLVJF","created_at":"2026-07-05T08:53:07.146508+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18664","citing_title":"NeuralMUSIC: A Hybrid Neural-Subspace Framework for Robot Sound Source Localization","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18664","citing_title":"NeuralMUSIC: A Hybrid Neural-Subspace Framework for Robot Sound Source Localization","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX","json":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX.json","graph_json":"https://pith.science/api/pith-number/JYZHLVJFOYKX2DPW4IBMZKM7OX/graph.json","events_json":"https://pith.science/api/pith-number/JYZHLVJFOYKX2DPW4IBMZKM7OX/events.json","paper":"https://pith.science/paper/JYZHLVJF"},"agent_actions":{"view_html":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX","download_json":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX.json","view_paper":"https://pith.science/paper/JYZHLVJF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.03927&json=true","fetch_graph":"https://pith.science/api/pith-number/JYZHLVJFOYKX2DPW4IBMZKM7OX/graph.json","fetch_events":"https://pith.science/api/pith-number/JYZHLVJFOYKX2DPW4IBMZKM7OX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX/action/storage_attestation","attest_author":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX/action/author_attestation","sign_citation":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX/action/citation_signature","submit_replication":"https://pith.science/pith/JYZHLVJFOYKX2DPW4IBMZKM7OX/action/replication_record"}},"created_at":"2026-07-05T08:53:07.146508+00:00","updated_at":"2026-07-05T08:53:07.146508+00:00"}