{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UGE2COYGSZ3SJQOF4FMZZFALQ6","short_pith_number":"pith:UGE2COYG","schema_version":"1.0","canonical_sha256":"a189a13b06967724c1c5e1599c940b8782df8646d50fc89baa3a7100a178fa12","source":{"kind":"arxiv","id":"2411.10193","version":2},"attestation_state":"computed","paper":{"title":"DiMoDif: Discourse Modality-information Differentiation for Audio-visual Deepfake Detection and Localization","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Christos Koutlis, Symeon Papadopoulos","submitted_at":"2024-11-15T13:47:33Z","abstract_excerpt":"Deepfake technology has rapidly advanced and poses significant threats to information integrity and trust in online multimedia. While significant progress has been made in detecting deepfakes, the simultaneous manipulation of audio and visual modalities, sometimes at small parts or in subtle ways, presents highly challenging detection scenarios. To address these challenges, we present DiMoDif, an audio-visual deepfake detection framework that leverages the inter-modality differences in machine perception of speech, based on the assumption that in real samples -- in contrast to deepfakes -- vis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.10193","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-15T13:47:33Z","cross_cats_sorted":[],"title_canon_sha256":"03d9317d719e479c0f747935939447c30a810e147b5b07f03814c5593ca2bee6","abstract_canon_sha256":"596ed70c5d5f272f09feba10a283743bd5d0a13c0be6d363fc5377aee0db5adf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:33.374470Z","signature_b64":"zZ/D9nUXb+FHnJ06B0Emnzes8rDwi4oEf13jK/iXNnRgH/G5STTc9r45Rly0s5LJLcJtGZCOQX8MXo79pto2CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a189a13b06967724c1c5e1599c940b8782df8646d50fc89baa3a7100a178fa12","last_reissued_at":"2026-07-05T10:47:33.373933Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:33.373933Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DiMoDif: Discourse Modality-information Differentiation for Audio-visual Deepfake Detection and Localization","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Christos Koutlis, Symeon Papadopoulos","submitted_at":"2024-11-15T13:47:33Z","abstract_excerpt":"Deepfake technology has rapidly advanced and poses significant threats to information integrity and trust in online multimedia. While significant progress has been made in detecting deepfakes, the simultaneous manipulation of audio and visual modalities, sometimes at small parts or in subtle ways, presents highly challenging detection scenarios. To address these challenges, we present DiMoDif, an audio-visual deepfake detection framework that leverages the inter-modality differences in machine perception of speech, based on the assumption that in real samples -- in contrast to deepfakes -- vis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.10193","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.10193/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.10193","created_at":"2026-07-05T10:47:33.373995+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.10193v2","created_at":"2026-07-05T10:47:33.373995+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.10193","created_at":"2026-07-05T10:47:33.373995+00:00"},{"alias_kind":"pith_short_12","alias_value":"UGE2COYGSZ3S","created_at":"2026-07-05T10:47:33.373995+00:00"},{"alias_kind":"pith_short_16","alias_value":"UGE2COYGSZ3SJQOF","created_at":"2026-07-05T10:47:33.373995+00:00"},{"alias_kind":"pith_short_8","alias_value":"UGE2COYG","created_at":"2026-07-05T10:47:33.373995+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00902","citing_title":"MG-RWKV: Multi-Grained Context-Aware RWKV for Temporal Forgery Localization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23113","citing_title":"Inconsistency-aware Multimodal Schr\\\"odinger Bridge for Deepfake Localization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09110","citing_title":"Generalizing Video DeepFake Detection by Self-generated Audio-Visual Pseudo-Fakes","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6","json":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6.json","graph_json":"https://pith.science/api/pith-number/UGE2COYGSZ3SJQOF4FMZZFALQ6/graph.json","events_json":"https://pith.science/api/pith-number/UGE2COYGSZ3SJQOF4FMZZFALQ6/events.json","paper":"https://pith.science/paper/UGE2COYG"},"agent_actions":{"view_html":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6","download_json":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6.json","view_paper":"https://pith.science/paper/UGE2COYG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.10193&json=true","fetch_graph":"https://pith.science/api/pith-number/UGE2COYGSZ3SJQOF4FMZZFALQ6/graph.json","fetch_events":"https://pith.science/api/pith-number/UGE2COYGSZ3SJQOF4FMZZFALQ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6/action/storage_attestation","attest_author":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6/action/author_attestation","sign_citation":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6/action/citation_signature","submit_replication":"https://pith.science/pith/UGE2COYGSZ3SJQOF4FMZZFALQ6/action/replication_record"}},"created_at":"2026-07-05T10:47:33.373995+00:00","updated_at":"2026-07-05T10:47:33.373995+00:00"}