{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:7FQDKLB4WDM63CRGMJ5X5XF5U7","short_pith_number":"pith:7FQDKLB4","schema_version":"1.0","canonical_sha256":"f960352c3cb0d9ed8a26627b7edcbda7d1e18a4726bad002c5ea37758bea3e82","source":{"kind":"arxiv","id":"2007.10937","version":1},"attestation_state":"computed","paper":{"title":"MovieNet: A Holistic Dataset for Movie Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anyi Rao, Dahua Lin, Jiaze Wang, Qingqiu Huang, Yu Xiong","submitted_at":"2020-07-21T16:54:33Z","abstract_excerpt":"Recent years have seen remarkable advances in visual understanding. However, how to understand a story-based long video with artistic styles, e.g. movie, remains challenging. In this paper, we introduce MovieNet -- a holistic dataset for movie understanding. MovieNet contains 1,100 movies with a large amount of multi-modal data, e.g. trailers, photos, plot descriptions, etc. Besides, different aspects of manual annotations are provided in MovieNet, including 1.1M characters with bounding boxes and identities, 42K scene boundaries, 2.5K aligned description sentences, 65K tags of place and actio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.10937","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2020-07-21T16:54:33Z","cross_cats_sorted":[],"title_canon_sha256":"a9430a62d1fe2df37a923d637abfaee3327e1543889ec6c8f45fc1a3d78a9dce","abstract_canon_sha256":"360544c3f57cfe5b4f01802c069ddcc2a017612a9f709598ab4346c68625c627"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:21:13.473248Z","signature_b64":"HxnmClVDwS/0Jv5YNn94aeVRrrUfi+jGy01ZVAw755os15NZHvjNu8BgtMViJrNw/vzXWe1vacU3DSAITF/0Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f960352c3cb0d9ed8a26627b7edcbda7d1e18a4726bad002c5ea37758bea3e82","last_reissued_at":"2026-07-05T01:21:13.472847Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:21:13.472847Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MovieNet: A Holistic Dataset for Movie Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anyi Rao, Dahua Lin, Jiaze Wang, Qingqiu Huang, Yu Xiong","submitted_at":"2020-07-21T16:54:33Z","abstract_excerpt":"Recent years have seen remarkable advances in visual understanding. However, how to understand a story-based long video with artistic styles, e.g. movie, remains challenging. In this paper, we introduce MovieNet -- a holistic dataset for movie understanding. MovieNet contains 1,100 movies with a large amount of multi-modal data, e.g. trailers, photos, plot descriptions, etc. Besides, different aspects of manual annotations are provided in MovieNet, including 1.1M characters with bounding boxes and identities, 42K scene boundaries, 2.5K aligned description sentences, 65K tags of place and actio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.10937","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.10937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.10937","created_at":"2026-07-05T01:21:13.472905+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.10937v1","created_at":"2026-07-05T01:21:13.472905+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.10937","created_at":"2026-07-05T01:21:13.472905+00:00"},{"alias_kind":"pith_short_12","alias_value":"7FQDKLB4WDM6","created_at":"2026-07-05T01:21:13.472905+00:00"},{"alias_kind":"pith_short_16","alias_value":"7FQDKLB4WDM63CRG","created_at":"2026-07-05T01:21:13.472905+00:00"},{"alias_kind":"pith_short_8","alias_value":"7FQDKLB4","created_at":"2026-07-05T01:21:13.472905+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.14724","citing_title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00181","citing_title":"CamReasoner: Reinforcing Camera Movement Understanding via Structured Spatial Reasoning","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7","json":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7.json","graph_json":"https://pith.science/api/pith-number/7FQDKLB4WDM63CRGMJ5X5XF5U7/graph.json","events_json":"https://pith.science/api/pith-number/7FQDKLB4WDM63CRGMJ5X5XF5U7/events.json","paper":"https://pith.science/paper/7FQDKLB4"},"agent_actions":{"view_html":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7","download_json":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7.json","view_paper":"https://pith.science/paper/7FQDKLB4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.10937&json=true","fetch_graph":"https://pith.science/api/pith-number/7FQDKLB4WDM63CRGMJ5X5XF5U7/graph.json","fetch_events":"https://pith.science/api/pith-number/7FQDKLB4WDM63CRGMJ5X5XF5U7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7/action/storage_attestation","attest_author":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7/action/author_attestation","sign_citation":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7/action/citation_signature","submit_replication":"https://pith.science/pith/7FQDKLB4WDM63CRGMJ5X5XF5U7/action/replication_record"}},"created_at":"2026-07-05T01:21:13.472905+00:00","updated_at":"2026-07-05T01:21:13.472905+00:00"}