{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3EKIREVPHRQFQIGVEAUQDN3RYG","short_pith_number":"pith:3EKIREVP","schema_version":"1.0","canonical_sha256":"d9148892af3c605820d5202901b771c1be1d19f611b9b08bbcd16bc0dcc812a4","source":{"kind":"arxiv","id":"2406.10221","version":2},"attestation_state":"computed","paper":{"title":"Long Story Short: Story-level Video Understanding from 20K Short Films","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Ivan Laptev, Ridouane Ghermi, Vicky Kalogeiton, Xi Wang","submitted_at":"2024-06-14T17:54:54Z","abstract_excerpt":"Recent developments in vision-language models have significantly advanced video understanding. Existing datasets and tasks, however, have notable limitations. Most datasets are confined to short videos with limited events and narrow narratives. For example, datasets with instructional and egocentric videos often depict activities of one person in a single scene. Although existing movie datasets offer richer content, they are often limited to short-term tasks, lack publicly available videos, and frequently encounter data leakage issues given the use of subtitles and other information about comm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10221","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-14T17:54:54Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"375a05379f05c0fc00731d7cf0aa35f6a882bd4762baaa492395a67eff9b97ef","abstract_canon_sha256":"29b7bbb919b6e419013e7108785fb1eae9013af67b256c1e95483049bed8c261"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:16.314773Z","signature_b64":"Ja7Vjn7OaJYbxEYwCNiCP9CFe+UB7hRu953WJ+xXR3T+IIW/wQ87GIuEutMA+KjfPeGtqEE0Md0l3oCl3BrdAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d9148892af3c605820d5202901b771c1be1d19f611b9b08bbcd16bc0dcc812a4","last_reissued_at":"2026-07-05T09:59:16.314220Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:16.314220Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long Story Short: Story-level Video Understanding from 20K Short Films","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Ivan Laptev, Ridouane Ghermi, Vicky Kalogeiton, Xi Wang","submitted_at":"2024-06-14T17:54:54Z","abstract_excerpt":"Recent developments in vision-language models have significantly advanced video understanding. Existing datasets and tasks, however, have notable limitations. Most datasets are confined to short videos with limited events and narrow narratives. For example, datasets with instructional and egocentric videos often depict activities of one person in a single scene. Although existing movie datasets offer richer content, they are often limited to short-term tasks, lack publicly available videos, and frequently encounter data leakage issues given the use of subtitles and other information about comm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10221","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10221","created_at":"2026-07-05T09:59:16.314290+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10221v2","created_at":"2026-07-05T09:59:16.314290+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10221","created_at":"2026-07-05T09:59:16.314290+00:00"},{"alias_kind":"pith_short_12","alias_value":"3EKIREVPHRQF","created_at":"2026-07-05T09:59:16.314290+00:00"},{"alias_kind":"pith_short_16","alias_value":"3EKIREVPHRQFQIGV","created_at":"2026-07-05T09:59:16.314290+00:00"},{"alias_kind":"pith_short_8","alias_value":"3EKIREVP","created_at":"2026-07-05T09:59:16.314290+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":277,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29473","citing_title":"MAVIN: Multi-Shot Audio-Visual Generation with Customized Narrative Control","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31069","citing_title":"Towards Effective Long-Video Event Prediction via Multi-Level Event Semantics Mining","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2501.02955","citing_title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18467","citing_title":"InstructAV2AV: Instruction-Guided Audio-Video Joint Editing","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2508.20765","citing_title":"Looking Beyond the Obvious: A Survey on Abstract Concept Recognition for Video Understanding","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10571","citing_title":"AVI-Edit: Audio-sync Video Instance Editing with Granularity-Aware Mask Refiner","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG","json":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG.json","graph_json":"https://pith.science/api/pith-number/3EKIREVPHRQFQIGVEAUQDN3RYG/graph.json","events_json":"https://pith.science/api/pith-number/3EKIREVPHRQFQIGVEAUQDN3RYG/events.json","paper":"https://pith.science/paper/3EKIREVP"},"agent_actions":{"view_html":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG","download_json":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG.json","view_paper":"https://pith.science/paper/3EKIREVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10221&json=true","fetch_graph":"https://pith.science/api/pith-number/3EKIREVPHRQFQIGVEAUQDN3RYG/graph.json","fetch_events":"https://pith.science/api/pith-number/3EKIREVPHRQFQIGVEAUQDN3RYG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG/action/storage_attestation","attest_author":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG/action/author_attestation","sign_citation":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG/action/citation_signature","submit_replication":"https://pith.science/pith/3EKIREVPHRQFQIGVEAUQDN3RYG/action/replication_record"}},"created_at":"2026-07-05T09:59:16.314290+00:00","updated_at":"2026-07-05T09:59:16.314290+00:00"}