{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JBCDGGJ4KSWX2U2EUWRLMMHY33","short_pith_number":"pith:JBCDGGJ4","schema_version":"1.0","canonical_sha256":"484433193c54ad7d5344a5a2b630f8deca2a3a1f1d5fb68c9d67836057d23311","source":{"kind":"arxiv","id":"2505.02013","version":1},"attestation_state":"computed","paper":{"title":"MLLM-Enhanced Face Forgery Detection: A Vision-Language Fusion Solution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ajian Liu, Haoyuan Zhang, Li Gao, Siran Peng, Tianshuo Zhang, Xiangyu Zhu, Zhen Lei, Zipei Wang","submitted_at":"2025-05-04T06:58:21Z","abstract_excerpt":"Reliable face forgery detection algorithms are crucial for countering the growing threat of deepfake-driven disinformation. Previous research has demonstrated the potential of Multimodal Large Language Models (MLLMs) in identifying manipulated faces. However, existing methods typically depend on either the Large Language Model (LLM) alone or an external detector to generate classification results, which often leads to sub-optimal integration of visual and textual modalities. In this paper, we propose VLF-FFD, a novel Vision-Language Fusion solution for MLLM-enhanced Face Forgery Detection. Our"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.02013","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-04T06:58:21Z","cross_cats_sorted":[],"title_canon_sha256":"3cf3b8f9cb3a30f2c08f99f3b87e67e3332354832465d54053092a52b721d4e6","abstract_canon_sha256":"f97b11ef11f26b4f9929d5a8969aec59cc285f203cf8a433a29885865b1aa4d6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:30.334098Z","signature_b64":"lg4zOvX185wcu+WvX4wlmIA/vnJuW/Bpvyt6RXWNEHIBb8x3y4uHW5t25ZjprdrT/C9yFIC21ZxldPj77d2eAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"484433193c54ad7d5344a5a2b630f8deca2a3a1f1d5fb68c9d67836057d23311","last_reissued_at":"2026-07-05T10:58:30.333557Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:30.333557Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLLM-Enhanced Face Forgery Detection: A Vision-Language Fusion Solution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ajian Liu, Haoyuan Zhang, Li Gao, Siran Peng, Tianshuo Zhang, Xiangyu Zhu, Zhen Lei, Zipei Wang","submitted_at":"2025-05-04T06:58:21Z","abstract_excerpt":"Reliable face forgery detection algorithms are crucial for countering the growing threat of deepfake-driven disinformation. Previous research has demonstrated the potential of Multimodal Large Language Models (MLLMs) in identifying manipulated faces. However, existing methods typically depend on either the Large Language Model (LLM) alone or an external detector to generate classification results, which often leads to sub-optimal integration of visual and textual modalities. In this paper, we propose VLF-FFD, a novel Vision-Language Fusion solution for MLLM-enhanced Face Forgery Detection. Our"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.02013","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.02013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.02013","created_at":"2026-07-05T10:58:30.333615+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.02013v1","created_at":"2026-07-05T10:58:30.333615+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.02013","created_at":"2026-07-05T10:58:30.333615+00:00"},{"alias_kind":"pith_short_12","alias_value":"JBCDGGJ4KSWX","created_at":"2026-07-05T10:58:30.333615+00:00"},{"alias_kind":"pith_short_16","alias_value":"JBCDGGJ4KSWX2U2E","created_at":"2026-07-05T10:58:30.333615+00:00"},{"alias_kind":"pith_short_8","alias_value":"JBCDGGJ4","created_at":"2026-07-05T10:58:30.333615+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.13660","citing_title":"VRAG-DFD: Verifiable Retrieval-Augmentation for MLLM-based Deepfake Detection","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33","json":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33.json","graph_json":"https://pith.science/api/pith-number/JBCDGGJ4KSWX2U2EUWRLMMHY33/graph.json","events_json":"https://pith.science/api/pith-number/JBCDGGJ4KSWX2U2EUWRLMMHY33/events.json","paper":"https://pith.science/paper/JBCDGGJ4"},"agent_actions":{"view_html":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33","download_json":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33.json","view_paper":"https://pith.science/paper/JBCDGGJ4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.02013&json=true","fetch_graph":"https://pith.science/api/pith-number/JBCDGGJ4KSWX2U2EUWRLMMHY33/graph.json","fetch_events":"https://pith.science/api/pith-number/JBCDGGJ4KSWX2U2EUWRLMMHY33/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33/action/storage_attestation","attest_author":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33/action/author_attestation","sign_citation":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33/action/citation_signature","submit_replication":"https://pith.science/pith/JBCDGGJ4KSWX2U2EUWRLMMHY33/action/replication_record"}},"created_at":"2026-07-05T10:58:30.333615+00:00","updated_at":"2026-07-05T10:58:30.333615+00:00"}