{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4GIKSUOVHFERP7M5RLFOXDZF6A","short_pith_number":"pith:4GIKSUOV","schema_version":"1.0","canonical_sha256":"e190a951d5394917fd9d8acaeb8f25f027c193a4c82228d1d2c8195c9e07af53","source":{"kind":"arxiv","id":"2407.12067","version":1},"attestation_state":"computed","paper":{"title":"MaskVD: Region Masking for Efficient Video Object Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chirayata Bhattacharyya, Gourav Datta, Kai Zheng, Peter A. Beerel, Souvik Kundu, Sreetama Sarkar","submitted_at":"2024-07-16T08:01:49Z","abstract_excerpt":"Video tasks are compute-heavy and thus pose a challenge when deploying in real-time applications, particularly for tasks that require state-of-the-art Vision Transformers (ViTs). Several research efforts have tried to address this challenge by leveraging the fact that large portions of the video undergo very little change across frames, leading to redundant computations in frame-based video processing. In particular, some works leverage pixel or semantic differences across frames, however, this yields limited latency benefits with significantly increased memory overhead. This paper, in contras"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12067","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-16T08:01:49Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b65c1d4bcc020386009127c5ef87ef392a824b1e582f8e52f0f5eceaf0f697e0","abstract_canon_sha256":"3e1a7a8481c5b6dadd5dc8424c92101db03fc5a0f667bdcd66de86f3d177678a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:59.621798Z","signature_b64":"LYOLiDoHXBSBg+rycXmsCRa0tWqo6ZvnjHUcfztWg9Y7/lKw8rHu5fGFijb4p9b2GFEX+sPlWFDxjfB2n5NdBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e190a951d5394917fd9d8acaeb8f25f027c193a4c82228d1d2c8195c9e07af53","last_reissued_at":"2026-07-05T08:44:59.621462Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:59.621462Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MaskVD: Region Masking for Efficient Video Object Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chirayata Bhattacharyya, Gourav Datta, Kai Zheng, Peter A. Beerel, Souvik Kundu, Sreetama Sarkar","submitted_at":"2024-07-16T08:01:49Z","abstract_excerpt":"Video tasks are compute-heavy and thus pose a challenge when deploying in real-time applications, particularly for tasks that require state-of-the-art Vision Transformers (ViTs). Several research efforts have tried to address this challenge by leveraging the fact that large portions of the video undergo very little change across frames, leading to redundant computations in frame-based video processing. In particular, some works leverage pixel or semantic differences across frames, however, this yields limited latency benefits with significantly increased memory overhead. This paper, in contras"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12067","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12067/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12067","created_at":"2026-07-05T08:44:59.621518+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12067v1","created_at":"2026-07-05T08:44:59.621518+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12067","created_at":"2026-07-05T08:44:59.621518+00:00"},{"alias_kind":"pith_short_12","alias_value":"4GIKSUOVHFER","created_at":"2026-07-05T08:44:59.621518+00:00"},{"alias_kind":"pith_short_16","alias_value":"4GIKSUOVHFERP7M5","created_at":"2026-07-05T08:44:59.621518+00:00"},{"alias_kind":"pith_short_8","alias_value":"4GIKSUOV","created_at":"2026-07-05T08:44:59.621518+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.05851","citing_title":"Temporal Cluster Assignment for Efficient Real-Time Video Segmentation","ref_index":37,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A","json":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A.json","graph_json":"https://pith.science/api/pith-number/4GIKSUOVHFERP7M5RLFOXDZF6A/graph.json","events_json":"https://pith.science/api/pith-number/4GIKSUOVHFERP7M5RLFOXDZF6A/events.json","paper":"https://pith.science/paper/4GIKSUOV"},"agent_actions":{"view_html":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A","download_json":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A.json","view_paper":"https://pith.science/paper/4GIKSUOV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12067&json=true","fetch_graph":"https://pith.science/api/pith-number/4GIKSUOVHFERP7M5RLFOXDZF6A/graph.json","fetch_events":"https://pith.science/api/pith-number/4GIKSUOVHFERP7M5RLFOXDZF6A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A/action/storage_attestation","attest_author":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A/action/author_attestation","sign_citation":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A/action/citation_signature","submit_replication":"https://pith.science/pith/4GIKSUOVHFERP7M5RLFOXDZF6A/action/replication_record"}},"created_at":"2026-07-05T08:44:59.621518+00:00","updated_at":"2026-07-05T08:44:59.621518+00:00"}