{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XLLCTAYID7GHZ4GC3FDT2KWMEI","short_pith_number":"pith:XLLCTAYI","schema_version":"1.0","canonical_sha256":"bad62983081fcc7cf0c2d9473d2acc223c5affc21597ba4ca396f106d5a3571b","source":{"kind":"arxiv","id":"2101.11080","version":1},"attestation_state":"computed","paper":{"title":"Deep Video Inpainting Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhinav Shrivastava, Larry S. Davis, Ning Yu, Peng Zhou, Ser-Nam Lim, Zuxuan Wu","submitted_at":"2021-01-26T20:53:49Z","abstract_excerpt":"This paper studies video inpainting detection, which localizes an inpainted region in a video both spatially and temporally. In particular, we introduce VIDNet, Video Inpainting Detection Network, which contains a two-stream encoder-decoder architecture with attention module. To reveal artifacts encoded in compression, VIDNet additionally takes in Error Level Analysis frames to augment RGB frames, producing multimodal features at different levels with an encoder. Exploring spatial and temporal relationships, these features are further decoded by a Convolutional LSTM to predict masks of inpaint"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.11080","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-01-26T20:53:49Z","cross_cats_sorted":[],"title_canon_sha256":"b1094dfd086b7e07fcb7dec06a67b455e320dd4fe3f00e8c6f8a8544485b9d31","abstract_canon_sha256":"d9f90f9776b5fef458286378abe728e90a31f22a0c4a8913c8a6f53d8adfb965"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:10:08.793154Z","signature_b64":"jfiWZSK7PlzcxbAFjPTuA243Y/oKO0md6KPWN4Aemk6hUgs5wrfhbipUJ/7IBzn2+D6aIFs1KOQ9N/+iZpkICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bad62983081fcc7cf0c2d9473d2acc223c5affc21597ba4ca396f106d5a3571b","last_reissued_at":"2026-07-05T02:10:08.792673Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:10:08.792673Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Video Inpainting Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhinav Shrivastava, Larry S. Davis, Ning Yu, Peng Zhou, Ser-Nam Lim, Zuxuan Wu","submitted_at":"2021-01-26T20:53:49Z","abstract_excerpt":"This paper studies video inpainting detection, which localizes an inpainted region in a video both spatially and temporally. In particular, we introduce VIDNet, Video Inpainting Detection Network, which contains a two-stream encoder-decoder architecture with attention module. To reveal artifacts encoded in compression, VIDNet additionally takes in Error Level Analysis frames to augment RGB frames, producing multimodal features at different levels with an encoder. Exploring spatial and temporal relationships, these features are further decoded by a Convolutional LSTM to predict masks of inpaint"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.11080","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.11080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.11080","created_at":"2026-07-05T02:10:08.792731+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.11080v1","created_at":"2026-07-05T02:10:08.792731+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.11080","created_at":"2026-07-05T02:10:08.792731+00:00"},{"alias_kind":"pith_short_12","alias_value":"XLLCTAYID7GH","created_at":"2026-07-05T02:10:08.792731+00:00"},{"alias_kind":"pith_short_16","alias_value":"XLLCTAYID7GHZ4GC","created_at":"2026-07-05T02:10:08.792731+00:00"},{"alias_kind":"pith_short_8","alias_value":"XLLCTAYI","created_at":"2026-07-05T02:10:08.792731+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.09459","citing_title":"RelayFormer: A Unified Local-Global Attention Framework for Scalable Image and Video Manipulation Localization","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI","json":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI.json","graph_json":"https://pith.science/api/pith-number/XLLCTAYID7GHZ4GC3FDT2KWMEI/graph.json","events_json":"https://pith.science/api/pith-number/XLLCTAYID7GHZ4GC3FDT2KWMEI/events.json","paper":"https://pith.science/paper/XLLCTAYI"},"agent_actions":{"view_html":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI","download_json":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI.json","view_paper":"https://pith.science/paper/XLLCTAYI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.11080&json=true","fetch_graph":"https://pith.science/api/pith-number/XLLCTAYID7GHZ4GC3FDT2KWMEI/graph.json","fetch_events":"https://pith.science/api/pith-number/XLLCTAYID7GHZ4GC3FDT2KWMEI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI/action/storage_attestation","attest_author":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI/action/author_attestation","sign_citation":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI/action/citation_signature","submit_replication":"https://pith.science/pith/XLLCTAYID7GHZ4GC3FDT2KWMEI/action/replication_record"}},"created_at":"2026-07-05T02:10:08.792731+00:00","updated_at":"2026-07-05T02:10:08.792731+00:00"}