{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XPJVJCLHB6ZPM3EJVPNF3CV7TP","short_pith_number":"pith:XPJVJCLH","schema_version":"1.0","canonical_sha256":"bbd35489670fb2f66c89abda5d8abf9bd7aa847f930c75067d6aef695efa7574","source":{"kind":"arxiv","id":"2108.07640","version":1},"attestation_state":"computed","paper":{"title":"Look Who's Talking: Active Speaker Detection in the Wild","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS","eess.IV"],"primary_cat":"cs.CV","authors_text":"Bong-Jin Lee, Hee-Soo Heo, Joon Son Chung, Soo-Whan Chung, Soyeon Choe, Yoohwan Kwon, You Jin Kim, Youngki Kwon","submitted_at":"2021-08-17T14:16:56Z","abstract_excerpt":"In this work, we present a novel audio-visual dataset for active speaker detection in the wild. A speaker is considered active when his or her face is visible and the voice is audible simultaneously. Although active speaker detection is a crucial pre-processing step for many audio-visual tasks, there is no existing dataset of natural human speech to evaluate the performance of active speaker detection. We therefore curate the Active Speakers in the Wild (ASW) dataset which contains videos and co-occurring speech segments with dense speech activity labels. Videos and timestamps of audible segme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.07640","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-08-17T14:16:56Z","cross_cats_sorted":["cs.SD","eess.AS","eess.IV"],"title_canon_sha256":"29371febe9d9fa9ce3650d656d2a27a1902c513f3037305c735983035a8415af","abstract_canon_sha256":"37a4370e0f2b72cbff1c9b49d652dd5ebf5a3a8cfb2e34ae22845b049db13991"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:06:41.138334Z","signature_b64":"ezMoh+zrHWNHmvds6g9HL7sKZh3dXnXNe0KqOI09WOYS+nm2D2Gs/mQ2k6Cwdl+70LoJeYoVHu2B88Q5o2UWDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bbd35489670fb2f66c89abda5d8abf9bd7aa847f930c75067d6aef695efa7574","last_reissued_at":"2026-07-05T03:06:41.137845Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:06:41.137845Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Look Who's Talking: Active Speaker Detection in the Wild","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS","eess.IV"],"primary_cat":"cs.CV","authors_text":"Bong-Jin Lee, Hee-Soo Heo, Joon Son Chung, Soo-Whan Chung, Soyeon Choe, Yoohwan Kwon, You Jin Kim, Youngki Kwon","submitted_at":"2021-08-17T14:16:56Z","abstract_excerpt":"In this work, we present a novel audio-visual dataset for active speaker detection in the wild. A speaker is considered active when his or her face is visible and the voice is audible simultaneously. Although active speaker detection is a crucial pre-processing step for many audio-visual tasks, there is no existing dataset of natural human speech to evaluate the performance of active speaker detection. We therefore curate the Active Speakers in the Wild (ASW) dataset which contains videos and co-occurring speech segments with dense speech activity labels. Videos and timestamps of audible segme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.07640","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.07640/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.07640","created_at":"2026-07-05T03:06:41.137901+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.07640v1","created_at":"2026-07-05T03:06:41.137901+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.07640","created_at":"2026-07-05T03:06:41.137901+00:00"},{"alias_kind":"pith_short_12","alias_value":"XPJVJCLHB6ZP","created_at":"2026-07-05T03:06:41.137901+00:00"},{"alias_kind":"pith_short_16","alias_value":"XPJVJCLHB6ZPM3EJ","created_at":"2026-07-05T03:06:41.137901+00:00"},{"alias_kind":"pith_short_8","alias_value":"XPJVJCLH","created_at":"2026-07-05T03:06:41.137901+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.02231","citing_title":"See, Hear, and Understand: Benchmarking Audiovisual Human Speech Understanding in Multimodal Large Language Models","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP","json":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP.json","graph_json":"https://pith.science/api/pith-number/XPJVJCLHB6ZPM3EJVPNF3CV7TP/graph.json","events_json":"https://pith.science/api/pith-number/XPJVJCLHB6ZPM3EJVPNF3CV7TP/events.json","paper":"https://pith.science/paper/XPJVJCLH"},"agent_actions":{"view_html":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP","download_json":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP.json","view_paper":"https://pith.science/paper/XPJVJCLH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.07640&json=true","fetch_graph":"https://pith.science/api/pith-number/XPJVJCLHB6ZPM3EJVPNF3CV7TP/graph.json","fetch_events":"https://pith.science/api/pith-number/XPJVJCLHB6ZPM3EJVPNF3CV7TP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP/action/storage_attestation","attest_author":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP/action/author_attestation","sign_citation":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP/action/citation_signature","submit_replication":"https://pith.science/pith/XPJVJCLHB6ZPM3EJVPNF3CV7TP/action/replication_record"}},"created_at":"2026-07-05T03:06:41.137901+00:00","updated_at":"2026-07-05T03:06:41.137901+00:00"}