{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:N5YNYBXMVTV7DVG7P5Q4BDANN5","short_pith_number":"pith:N5YNYBXM","schema_version":"1.0","canonical_sha256":"6f70dc06ecacebf1d4df7f61c08c0d6f5399ae2b726f595e7af31084fe40b85c","source":{"kind":"arxiv","id":"2309.16569","version":1},"attestation_state":"computed","paper":{"title":"Audio-Visual Speaker Verification via Joint Cross-Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jahangir Alam, R. Gnana Praveen","submitted_at":"2023-09-28T16:25:29Z","abstract_excerpt":"Speaker verification has been widely explored using speech signals, which has shown significant improvement using deep models. Recently, there has been a surge in exploring faces and voices as they can offer more complementary and comprehensive information than relying only on a single modality of speech signals. Though current methods in the literature on the fusion of faces and voices have shown improvement over that of individual face or voice modalities, the potential of audio-visual fusion is not fully explored for speaker verification. Most of the existing methods based on audio-visual f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16569","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2023-09-28T16:25:29Z","cross_cats_sorted":["cs.CV","cs.MM","eess.AS"],"title_canon_sha256":"89fc4cbd0bcce3b5c9f95ea8b7feee3570e87991fae625af072d5529e10a6f6a","abstract_canon_sha256":"8fb8fed3fa70b4094d3830ef07769f356f2cbd65ef305838740bc17ac5ad0233"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:18.417880Z","signature_b64":"YDMM07wCmhDHxBYYAAB91QcG5HQ9t2oArIu2CVGKpSfznIVA6+MUuFRmuX20FnGpwQlxacRknwipgzXA+qQADA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f70dc06ecacebf1d4df7f61c08c0d6f5399ae2b726f595e7af31084fe40b85c","last_reissued_at":"2026-07-05T06:55:18.417391Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:18.417391Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio-Visual Speaker Verification via Joint Cross-Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Jahangir Alam, R. Gnana Praveen","submitted_at":"2023-09-28T16:25:29Z","abstract_excerpt":"Speaker verification has been widely explored using speech signals, which has shown significant improvement using deep models. Recently, there has been a surge in exploring faces and voices as they can offer more complementary and comprehensive information than relying only on a single modality of speech signals. Though current methods in the literature on the fusion of faces and voices have shown improvement over that of individual face or voice modalities, the potential of audio-visual fusion is not fully explored for speaker verification. Most of the existing methods based on audio-visual f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16569","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16569/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16569","created_at":"2026-07-05T06:55:18.417451+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16569v1","created_at":"2026-07-05T06:55:18.417451+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16569","created_at":"2026-07-05T06:55:18.417451+00:00"},{"alias_kind":"pith_short_12","alias_value":"N5YNYBXMVTV7","created_at":"2026-07-05T06:55:18.417451+00:00"},{"alias_kind":"pith_short_16","alias_value":"N5YNYBXMVTV7DVG7","created_at":"2026-07-05T06:55:18.417451+00:00"},{"alias_kind":"pith_short_8","alias_value":"N5YNYBXM","created_at":"2026-07-05T06:55:18.417451+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12495","citing_title":"Missing-Token Prompted Reliability-Aware Fusion for Robust Polyglot Speaker Identification","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5","json":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5.json","graph_json":"https://pith.science/api/pith-number/N5YNYBXMVTV7DVG7P5Q4BDANN5/graph.json","events_json":"https://pith.science/api/pith-number/N5YNYBXMVTV7DVG7P5Q4BDANN5/events.json","paper":"https://pith.science/paper/N5YNYBXM"},"agent_actions":{"view_html":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5","download_json":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5.json","view_paper":"https://pith.science/paper/N5YNYBXM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16569&json=true","fetch_graph":"https://pith.science/api/pith-number/N5YNYBXMVTV7DVG7P5Q4BDANN5/graph.json","fetch_events":"https://pith.science/api/pith-number/N5YNYBXMVTV7DVG7P5Q4BDANN5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5/action/storage_attestation","attest_author":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5/action/author_attestation","sign_citation":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5/action/citation_signature","submit_replication":"https://pith.science/pith/N5YNYBXMVTV7DVG7P5Q4BDANN5/action/replication_record"}},"created_at":"2026-07-05T06:55:18.417451+00:00","updated_at":"2026-07-05T06:55:18.417451+00:00"}