{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OKBOS3BJLBYDC75LLFP3JWWLJG","short_pith_number":"pith:OKBOS3BJ","schema_version":"1.0","canonical_sha256":"7282e96c295870317fab595fb4dacb499d7fc602d2cf3908fdd8cd95263dc5af","source":{"kind":"arxiv","id":"2306.12106","version":2},"attestation_state":"computed","paper":{"title":"ViTEraser: Harnessing the Power of Vision Transformers for Scene Text Removal with SegMIM Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongyu Liu, Dezhi Peng, Lianwen Jin, Yuliang Liu","submitted_at":"2023-06-21T08:47:20Z","abstract_excerpt":"Scene text removal (STR) aims at replacing text strokes in natural scenes with visually coherent backgrounds. Recent STR approaches rely on iterative refinements or explicit text masks, resulting in high complexity and sensitivity to the accuracy of text localization. Moreover, most existing STR methods adopt convolutional architectures while the potential of vision Transformers (ViTs) remains largely unexplored. In this paper, we propose a simple-yet-effective ViT-based text eraser, dubbed ViTEraser. Following a concise encoder-decoder framework, ViTEraser can easily incorporate various ViTs "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.12106","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-06-21T08:47:20Z","cross_cats_sorted":[],"title_canon_sha256":"6b2d133c2224f5f23a1cdf20daa3f35bb10241a3730d8060282465a8959d950f","abstract_canon_sha256":"e197b78eaf31f25feaabe41a87ad074662299ba7244615db1be10956f03a5f50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:46:19.717776Z","signature_b64":"SWjaTjpumuDhS38b/hXe2SSZAeH1lXRR4GVMCW/jmBW0zcuB7BVmNtwwWXXWs4NfUcapeDrvNpDBfwA+1gQzCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7282e96c295870317fab595fb4dacb499d7fc602d2cf3908fdd8cd95263dc5af","last_reissued_at":"2026-07-05T07:46:19.717291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:46:19.717291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViTEraser: Harnessing the Power of Vision Transformers for Scene Text Removal with SegMIM Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongyu Liu, Dezhi Peng, Lianwen Jin, Yuliang Liu","submitted_at":"2023-06-21T08:47:20Z","abstract_excerpt":"Scene text removal (STR) aims at replacing text strokes in natural scenes with visually coherent backgrounds. Recent STR approaches rely on iterative refinements or explicit text masks, resulting in high complexity and sensitivity to the accuracy of text localization. Moreover, most existing STR methods adopt convolutional architectures while the potential of vision Transformers (ViTs) remains largely unexplored. In this paper, we propose a simple-yet-effective ViT-based text eraser, dubbed ViTEraser. Following a concise encoder-decoder framework, ViTEraser can easily incorporate various ViTs "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.12106","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.12106/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.12106","created_at":"2026-07-05T07:46:19.717359+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.12106v2","created_at":"2026-07-05T07:46:19.717359+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.12106","created_at":"2026-07-05T07:46:19.717359+00:00"},{"alias_kind":"pith_short_12","alias_value":"OKBOS3BJLBYD","created_at":"2026-07-05T07:46:19.717359+00:00"},{"alias_kind":"pith_short_16","alias_value":"OKBOS3BJLBYDC75L","created_at":"2026-07-05T07:46:19.717359+00:00"},{"alias_kind":"pith_short_8","alias_value":"OKBOS3BJ","created_at":"2026-07-05T07:46:19.717359+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG","json":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG.json","graph_json":"https://pith.science/api/pith-number/OKBOS3BJLBYDC75LLFP3JWWLJG/graph.json","events_json":"https://pith.science/api/pith-number/OKBOS3BJLBYDC75LLFP3JWWLJG/events.json","paper":"https://pith.science/paper/OKBOS3BJ"},"agent_actions":{"view_html":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG","download_json":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG.json","view_paper":"https://pith.science/paper/OKBOS3BJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.12106&json=true","fetch_graph":"https://pith.science/api/pith-number/OKBOS3BJLBYDC75LLFP3JWWLJG/graph.json","fetch_events":"https://pith.science/api/pith-number/OKBOS3BJLBYDC75LLFP3JWWLJG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG/action/storage_attestation","attest_author":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG/action/author_attestation","sign_citation":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG/action/citation_signature","submit_replication":"https://pith.science/pith/OKBOS3BJLBYDC75LLFP3JWWLJG/action/replication_record"}},"created_at":"2026-07-05T07:46:19.717359+00:00","updated_at":"2026-07-05T07:46:19.717359+00:00"}