{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LMQMGOGGUUDWJHZX2GJG66ZS2X","short_pith_number":"pith:LMQMGOGG","schema_version":"1.0","canonical_sha256":"5b20c338c6a507649f37d1926f7b32d5e884de368b2106d72f4b8dd459128a34","source":{"kind":"arxiv","id":"2506.08372","version":1},"attestation_state":"computed","paper":{"title":"Multimodal Zero-Shot Framework for Deepfake Hate Speech Detection in Low-Resource Languages","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Likhith Ayinala, Mayank Vatsa, Richa Singh, Rishabh Ranjan","submitted_at":"2025-06-10T02:37:42Z","abstract_excerpt":"This paper introduces a novel multimodal framework for hate speech detection in deepfake audio, excelling even in zero-shot scenarios. Unlike previous approaches, our method uses contrastive learning to jointly align audio and text representations across languages. We present the first benchmark dataset with 127,290 paired text and synthesized speech samples in six languages: English and five low-resource Indian languages (Hindi, Bengali, Marathi, Tamil, Telugu). Our model learns a shared semantic embedding space, enabling robust cross-lingual and cross-modal classification. Experiments on two"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.08372","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2025-06-10T02:37:42Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"324dcc03013381947063500a97974d06099ef6fd9f2d8bcab3d037725100ea04","abstract_canon_sha256":"de3193718e3c0542f3df3ee258d74b55e626f78227bd1d3f31fd7d5dd57e384b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:09.264398Z","signature_b64":"dl4BzMPIoEllctgu4Z19aev+gWWN7ZzR8OHjHtZyG7Ol4ERu1lVIBgKfMKZqvHDDfJ/UC0F5scmcNjevU2TwCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b20c338c6a507649f37d1926f7b32d5e884de368b2106d72f4b8dd459128a34","last_reissued_at":"2026-07-05T11:19:09.263902Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:09.263902Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Zero-Shot Framework for Deepfake Hate Speech Detection in Low-Resource Languages","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Likhith Ayinala, Mayank Vatsa, Richa Singh, Rishabh Ranjan","submitted_at":"2025-06-10T02:37:42Z","abstract_excerpt":"This paper introduces a novel multimodal framework for hate speech detection in deepfake audio, excelling even in zero-shot scenarios. Unlike previous approaches, our method uses contrastive learning to jointly align audio and text representations across languages. We present the first benchmark dataset with 127,290 paired text and synthesized speech samples in six languages: English and five low-resource Indian languages (Hindi, Bengali, Marathi, Tamil, Telugu). Our model learns a shared semantic embedding space, enabling robust cross-lingual and cross-modal classification. Experiments on two"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.08372","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.08372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.08372","created_at":"2026-07-05T11:19:09.263957+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.08372v1","created_at":"2026-07-05T11:19:09.263957+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.08372","created_at":"2026-07-05T11:19:09.263957+00:00"},{"alias_kind":"pith_short_12","alias_value":"LMQMGOGGUUDW","created_at":"2026-07-05T11:19:09.263957+00:00"},{"alias_kind":"pith_short_16","alias_value":"LMQMGOGGUUDWJHZX","created_at":"2026-07-05T11:19:09.263957+00:00"},{"alias_kind":"pith_short_8","alias_value":"LMQMGOGG","created_at":"2026-07-05T11:19:09.263957+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.08372","citing_title":"Multimodal Zero-Shot Framework for Deepfake Hate Speech Detection in Low-Resource Languages","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X","json":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X.json","graph_json":"https://pith.science/api/pith-number/LMQMGOGGUUDWJHZX2GJG66ZS2X/graph.json","events_json":"https://pith.science/api/pith-number/LMQMGOGGUUDWJHZX2GJG66ZS2X/events.json","paper":"https://pith.science/paper/LMQMGOGG"},"agent_actions":{"view_html":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X","download_json":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X.json","view_paper":"https://pith.science/paper/LMQMGOGG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.08372&json=true","fetch_graph":"https://pith.science/api/pith-number/LMQMGOGGUUDWJHZX2GJG66ZS2X/graph.json","fetch_events":"https://pith.science/api/pith-number/LMQMGOGGUUDWJHZX2GJG66ZS2X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X/action/storage_attestation","attest_author":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X/action/author_attestation","sign_citation":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X/action/citation_signature","submit_replication":"https://pith.science/pith/LMQMGOGGUUDWJHZX2GJG66ZS2X/action/replication_record"}},"created_at":"2026-07-05T11:19:09.263957+00:00","updated_at":"2026-07-05T11:19:09.263957+00:00"}