{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3XMWQD5D65E4ZX4HF44SSTXUVX","short_pith_number":"pith:3XMWQD5D","schema_version":"1.0","canonical_sha256":"ddd9680fa3f749ccdf872f39294ef4adcb6738819ee4d35df1f82df55888fd22","source":{"kind":"arxiv","id":"2406.06965","version":4},"attestation_state":"computed","paper":{"title":"Evolving from Single-modal to Multi-modal Facial Deepfake Detection: Progress and Challenges","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Joey Tianyi Zhou, Ping Liu, Qiqi Tao","submitted_at":"2024-06-11T05:48:04Z","abstract_excerpt":"As synthetic media, including video, audio, and text, become increasingly indistinguishable from real content, the risks of misinformation, identity fraud, and social manipulation escalate. This survey traces the evolution of deepfake detection from early single-modal methods to sophisticated multi-modal approaches that integrate audio-visual and text-visual cues. We present a structured taxonomy of detection techniques and analyze the transition from GAN-based to diffusion model-driven deepfakes, which introduce new challenges due to their heightened realism and robustness against detection. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.06965","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-11T05:48:04Z","cross_cats_sorted":[],"title_canon_sha256":"dcce5b46a15f88e8fd2e9dc09a0d002f66a74b2eedaca1c026707f8d5eb1bf61","abstract_canon_sha256":"1da12d0a824438209a151059f5af8d8f1450465e4714ad59338fc2cd552fdb5b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:36.225188Z","signature_b64":"Zvf3h0VEiFoHasIkn/bjU+NiILpAz8w9o3YOCsDNZ1mR30KmpHxje5S/3z6aSKilcPsKjekJ5bs9cGaD4yhWAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddd9680fa3f749ccdf872f39294ef4adcb6738819ee4d35df1f82df55888fd22","last_reissued_at":"2026-07-05T10:43:36.224691Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:36.224691Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evolving from Single-modal to Multi-modal Facial Deepfake Detection: Progress and Challenges","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Joey Tianyi Zhou, Ping Liu, Qiqi Tao","submitted_at":"2024-06-11T05:48:04Z","abstract_excerpt":"As synthetic media, including video, audio, and text, become increasingly indistinguishable from real content, the risks of misinformation, identity fraud, and social manipulation escalate. This survey traces the evolution of deepfake detection from early single-modal methods to sophisticated multi-modal approaches that integrate audio-visual and text-visual cues. We present a structured taxonomy of detection techniques and analyze the transition from GAN-based to diffusion model-driven deepfakes, which introduce new challenges due to their heightened realism and robustness against detection. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.06965","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.06965/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.06965","created_at":"2026-07-05T10:43:36.224743+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.06965v4","created_at":"2026-07-05T10:43:36.224743+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.06965","created_at":"2026-07-05T10:43:36.224743+00:00"},{"alias_kind":"pith_short_12","alias_value":"3XMWQD5D65E4","created_at":"2026-07-05T10:43:36.224743+00:00"},{"alias_kind":"pith_short_16","alias_value":"3XMWQD5D65E4ZX4H","created_at":"2026-07-05T10:43:36.224743+00:00"},{"alias_kind":"pith_short_8","alias_value":"3XMWQD5D","created_at":"2026-07-05T10:43:36.224743+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17573","citing_title":"Deepfake Detection in Social Media: A Temporal Artifact Analysis Using 3D Convolutional Neural Networks","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28022","citing_title":"Are DeepFakes Realistic Enough? Exploring Semantic Mismatch as a Novel Challenge","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10522","citing_title":"SEED: A Large-Scale Benchmark for Provenance Tracing in Sequential Deepfake Facial Edits","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14570","citing_title":"Deepfake Detection Generalization with Diffusion Noise","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX","json":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX.json","graph_json":"https://pith.science/api/pith-number/3XMWQD5D65E4ZX4HF44SSTXUVX/graph.json","events_json":"https://pith.science/api/pith-number/3XMWQD5D65E4ZX4HF44SSTXUVX/events.json","paper":"https://pith.science/paper/3XMWQD5D"},"agent_actions":{"view_html":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX","download_json":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX.json","view_paper":"https://pith.science/paper/3XMWQD5D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.06965&json=true","fetch_graph":"https://pith.science/api/pith-number/3XMWQD5D65E4ZX4HF44SSTXUVX/graph.json","fetch_events":"https://pith.science/api/pith-number/3XMWQD5D65E4ZX4HF44SSTXUVX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX/action/storage_attestation","attest_author":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX/action/author_attestation","sign_citation":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX/action/citation_signature","submit_replication":"https://pith.science/pith/3XMWQD5D65E4ZX4HF44SSTXUVX/action/replication_record"}},"created_at":"2026-07-05T10:43:36.224743+00:00","updated_at":"2026-07-05T10:43:36.224743+00:00"}