{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6P6JWWP4MQC4JHMYFC5MW2IKSK","short_pith_number":"pith:6P6JWWP4","schema_version":"1.0","canonical_sha256":"f3fc9b59fc6405c49d9828bacb690a929bca6f91752fbff313b146f854a3a950","source":{"kind":"arxiv","id":"2509.10278","version":1},"attestation_state":"computed","paper":{"title":"Detecting Text Manipulation in Images using Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Amir Mohammadi, Christophe Ecabert, Ketan Kotwal, Pavel Korshunov, S\\'ebastien Marcel, Vidit Vidit","submitted_at":"2025-09-12T14:20:29Z","abstract_excerpt":"Recent works have shown the effectiveness of Large Vision Language Models (VLMs or LVLMs) in image manipulation detection. However, text manipulation detection is largely missing in these studies. We bridge this knowledge gap by analyzing closed- and open-source VLMs on different text manipulation datasets. Our results suggest that open-source models are getting closer, but still behind closed-source ones like GPT- 4o. Additionally, we benchmark image manipulation detection-specific VLMs for text manipulation detection and show that they suffer from the generalization problem. We benchmark VLM"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.10278","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-09-12T14:20:29Z","cross_cats_sorted":[],"title_canon_sha256":"a0417a3a3b1d5e0216c7dc0f18115cfb1f70ce37b54fbb5daa42107c90257965","abstract_canon_sha256":"b4781cd2eeaf5da23f9eb2de1d6d3682fa6f012febd6a82d577e253310102d3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:06.228790Z","signature_b64":"+FU+IhxmhSX7kyVqKiWAR+JVohiu8UtGDk66bkAw/IwdtlIZmPBVKkSvlQulcvfbdKZ8HTd1X0shrrn6RtDEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3fc9b59fc6405c49d9828bacb690a929bca6f91752fbff313b146f854a3a950","last_reissued_at":"2026-07-05T12:11:06.228261Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:06.228261Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Detecting Text Manipulation in Images using Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Amir Mohammadi, Christophe Ecabert, Ketan Kotwal, Pavel Korshunov, S\\'ebastien Marcel, Vidit Vidit","submitted_at":"2025-09-12T14:20:29Z","abstract_excerpt":"Recent works have shown the effectiveness of Large Vision Language Models (VLMs or LVLMs) in image manipulation detection. However, text manipulation detection is largely missing in these studies. We bridge this knowledge gap by analyzing closed- and open-source VLMs on different text manipulation datasets. Our results suggest that open-source models are getting closer, but still behind closed-source ones like GPT- 4o. Additionally, we benchmark image manipulation detection-specific VLMs for text manipulation detection and show that they suffer from the generalization problem. We benchmark VLM"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.10278","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.10278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.10278","created_at":"2026-07-05T12:11:06.228326+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.10278v1","created_at":"2026-07-05T12:11:06.228326+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.10278","created_at":"2026-07-05T12:11:06.228326+00:00"},{"alias_kind":"pith_short_12","alias_value":"6P6JWWP4MQC4","created_at":"2026-07-05T12:11:06.228326+00:00"},{"alias_kind":"pith_short_16","alias_value":"6P6JWWP4MQC4JHMY","created_at":"2026-07-05T12:11:06.228326+00:00"},{"alias_kind":"pith_short_8","alias_value":"6P6JWWP4","created_at":"2026-07-05T12:11:06.228326+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01442","citing_title":"From Forgeries to Foundation Models: A Systematic Survey of Identity Document Attack and Detection","ref_index":98,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK","json":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK.json","graph_json":"https://pith.science/api/pith-number/6P6JWWP4MQC4JHMYFC5MW2IKSK/graph.json","events_json":"https://pith.science/api/pith-number/6P6JWWP4MQC4JHMYFC5MW2IKSK/events.json","paper":"https://pith.science/paper/6P6JWWP4"},"agent_actions":{"view_html":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK","download_json":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK.json","view_paper":"https://pith.science/paper/6P6JWWP4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.10278&json=true","fetch_graph":"https://pith.science/api/pith-number/6P6JWWP4MQC4JHMYFC5MW2IKSK/graph.json","fetch_events":"https://pith.science/api/pith-number/6P6JWWP4MQC4JHMYFC5MW2IKSK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK/action/storage_attestation","attest_author":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK/action/author_attestation","sign_citation":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK/action/citation_signature","submit_replication":"https://pith.science/pith/6P6JWWP4MQC4JHMYFC5MW2IKSK/action/replication_record"}},"created_at":"2026-07-05T12:11:06.228326+00:00","updated_at":"2026-07-05T12:11:06.228326+00:00"}