{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SS3QO6AGL33UBC5JMLSJXFEMR2","short_pith_number":"pith:SS3QO6AG","schema_version":"1.0","canonical_sha256":"94b70778065ef7408ba962e49b948c8e9d0da6745c9008f19a5ea27af04c5c72","source":{"kind":"arxiv","id":"2410.14509","version":1},"attestation_state":"computed","paper":{"title":"CLIP-VAD: Exploiting Vision-Language Models for Voice Activity Detection","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrea Appiani, Cigdem Beyan","submitted_at":"2024-10-18T14:43:34Z","abstract_excerpt":"Voice Activity Detection (VAD) is the process of automatically determining whether a person is speaking and identifying the timing of their speech in an audiovisual data. Traditionally, this task has been tackled by processing either audio signals or visual data, or by combining both modalities through fusion or joint learning. In our study, drawing inspiration from recent advancements in visual-language models, we introduce a novel approach leveraging Contrastive Language-Image Pretraining (CLIP) models. The CLIP visual encoder analyzes video segments composed of the upper body of an individu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14509","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-18T14:43:34Z","cross_cats_sorted":[],"title_canon_sha256":"1d66787ea1bbece6afdf1e691726a55ca27a5d154094016d641d3a8497976bdd","abstract_canon_sha256":"9421bb6908fee87c45e0bc73fa2a4679239ec399e34ccf6bc9df677f796e4fd6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:33.339679Z","signature_b64":"67RbUCLZMztUnInG0FqjlShgmY64mAMnM0en9xPUXjXWEqFd1zwMW5nx9xCXa6xqYunqX6/GC0r8m0PnBa6YBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"94b70778065ef7408ba962e49b948c8e9d0da6745c9008f19a5ea27af04c5c72","last_reissued_at":"2026-07-05T09:22:33.339192Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:33.339192Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIP-VAD: Exploiting Vision-Language Models for Voice Activity Detection","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrea Appiani, Cigdem Beyan","submitted_at":"2024-10-18T14:43:34Z","abstract_excerpt":"Voice Activity Detection (VAD) is the process of automatically determining whether a person is speaking and identifying the timing of their speech in an audiovisual data. Traditionally, this task has been tackled by processing either audio signals or visual data, or by combining both modalities through fusion or joint learning. In our study, drawing inspiration from recent advancements in visual-language models, we introduce a novel approach leveraging Contrastive Language-Image Pretraining (CLIP) models. The CLIP visual encoder analyzes video segments composed of the upper body of an individu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14509","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14509/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14509","created_at":"2026-07-05T09:22:33.339245+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14509v1","created_at":"2026-07-05T09:22:33.339245+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14509","created_at":"2026-07-05T09:22:33.339245+00:00"},{"alias_kind":"pith_short_12","alias_value":"SS3QO6AGL33U","created_at":"2026-07-05T09:22:33.339245+00:00"},{"alias_kind":"pith_short_16","alias_value":"SS3QO6AGL33UBC5J","created_at":"2026-07-05T09:22:33.339245+00:00"},{"alias_kind":"pith_short_8","alias_value":"SS3QO6AG","created_at":"2026-07-05T09:22:33.339245+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.20885","citing_title":"SincQDR-VAD: A Noise-Robust Voice Activity Detection Framework Leveraging Learnable Filters and Ranking-Aware Optimization","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2","json":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2.json","graph_json":"https://pith.science/api/pith-number/SS3QO6AGL33UBC5JMLSJXFEMR2/graph.json","events_json":"https://pith.science/api/pith-number/SS3QO6AGL33UBC5JMLSJXFEMR2/events.json","paper":"https://pith.science/paper/SS3QO6AG"},"agent_actions":{"view_html":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2","download_json":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2.json","view_paper":"https://pith.science/paper/SS3QO6AG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14509&json=true","fetch_graph":"https://pith.science/api/pith-number/SS3QO6AGL33UBC5JMLSJXFEMR2/graph.json","fetch_events":"https://pith.science/api/pith-number/SS3QO6AGL33UBC5JMLSJXFEMR2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2/action/storage_attestation","attest_author":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2/action/author_attestation","sign_citation":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2/action/citation_signature","submit_replication":"https://pith.science/pith/SS3QO6AGL33UBC5JMLSJXFEMR2/action/replication_record"}},"created_at":"2026-07-05T09:22:33.339245+00:00","updated_at":"2026-07-05T09:22:33.339245+00:00"}