{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6QGR4UNH46SL5DYRCSFXQEWHJB","short_pith_number":"pith:6QGR4UNH","schema_version":"1.0","canonical_sha256":"f40d1e51a7e7a4be8f11148b7812c7486190dcc9af7b126d58bca0b95026086d","source":{"kind":"arxiv","id":"2507.20939","version":1},"attestation_state":"computed","paper":{"title":"ARC-Hunyuan-Video-7B: Structured Video Comprehension of Real-World Shorts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Li, Di Wang, Han Hu, Jin Ma, Jinwen Luo, Junfu Pu, Lisheng Duan, Lu Qiu, Teng Wang, Weibo Gu, Xiaojing Zhang, Xinyu Zuo, Yangyu Tao, Ying Shan, Yixiao Ge, Yizhuo Li, Yuying Ge, Zexuan Li","submitted_at":"2025-07-28T15:52:36Z","abstract_excerpt":"Real-world user-generated short videos, especially those distributed on platforms such as WeChat Channel and TikTok, dominate the mobile internet. However, current large multimodal models lack essential temporally-structured, detailed, and in-depth video comprehension capabilities, which are the cornerstone of effective video search and recommendation, as well as emerging video applications. Understanding real-world shorts is actually challenging due to their complex visual elements, high information density in both visuals and audio, and fast pacing that focuses on emotional expression and vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.20939","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-28T15:52:36Z","cross_cats_sorted":[],"title_canon_sha256":"3e8d1d8a4afb082ad98e56356031df6ae1d6abba920e4465ca1cd58da67ac1d6","abstract_canon_sha256":"c542f8ba051cfde34a0d66dcd4db1af5aa44c93961ddda8d0161b552fe674d40"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:28.605108Z","signature_b64":"u4RTDoRoXHFhXqZ80BfXfymz/4oZB+bq/l8CnvP4ADhSDnCYNOfQoYpKkZ826V1jArj7XoRHjbcKtUnDvS8kCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f40d1e51a7e7a4be8f11148b7812c7486190dcc9af7b126d58bca0b95026086d","last_reissued_at":"2026-07-05T11:44:28.604634Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:28.604634Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ARC-Hunyuan-Video-7B: Structured Video Comprehension of Real-World Shorts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Li, Di Wang, Han Hu, Jin Ma, Jinwen Luo, Junfu Pu, Lisheng Duan, Lu Qiu, Teng Wang, Weibo Gu, Xiaojing Zhang, Xinyu Zuo, Yangyu Tao, Ying Shan, Yixiao Ge, Yizhuo Li, Yuying Ge, Zexuan Li","submitted_at":"2025-07-28T15:52:36Z","abstract_excerpt":"Real-world user-generated short videos, especially those distributed on platforms such as WeChat Channel and TikTok, dominate the mobile internet. However, current large multimodal models lack essential temporally-structured, detailed, and in-depth video comprehension capabilities, which are the cornerstone of effective video search and recommendation, as well as emerging video applications. Understanding real-world shorts is actually challenging due to their complex visual elements, high information density in both visuals and audio, and fast pacing that focuses on emotional expression and vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.20939","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.20939/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.20939","created_at":"2026-07-05T11:44:28.604681+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.20939v1","created_at":"2026-07-05T11:44:28.604681+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.20939","created_at":"2026-07-05T11:44:28.604681+00:00"},{"alias_kind":"pith_short_12","alias_value":"6QGR4UNH46SL","created_at":"2026-07-05T11:44:28.604681+00:00"},{"alias_kind":"pith_short_16","alias_value":"6QGR4UNH46SL5DYR","created_at":"2026-07-05T11:44:28.604681+00:00"},{"alias_kind":"pith_short_8","alias_value":"6QGR4UNH","created_at":"2026-07-05T11:44:28.604681+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20970","citing_title":"CogniRoute: Learning to Route Social Evidence in Omni-Modal Models","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01667","citing_title":"Temporal and Cross-Modal Alignment for Enhanced Audiovisual Video Captioning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31187","citing_title":"Learning to Deny: Action Denial in Multimodal Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26584","citing_title":"O-MARC: Omni Memory-Augmented Compression Distillation for Efficient Video Understanding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20035","citing_title":"Stage-adaptive Token Selection for Efficient Omni-modal LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14582","citing_title":"OmniZip: Audio-Guided Dynamic Token Compression for Fast Omnimodal Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.16918","citing_title":"AdaTooler-V: Adaptive Tool-Use for Images and Videos","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21334","citing_title":"Streaming Video Instruction Tuning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02994","citing_title":"Video-OPD: Efficient Post-Training of Multimodal Large Language Models for Temporal Video Grounding via On-Policy Distillation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12056","citing_title":"OmniRefine: Alignment-Aware Cooperative Compression for Efficient Omnimodal Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23198","citing_title":"StoryTR: Narrative-Centric Video Temporal Retrieval with Theory of Mind Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11102","citing_title":"OmniScript: Towards Audio-Visual Script Generation for Long-Form Cinematic Video","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB","json":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB.json","graph_json":"https://pith.science/api/pith-number/6QGR4UNH46SL5DYRCSFXQEWHJB/graph.json","events_json":"https://pith.science/api/pith-number/6QGR4UNH46SL5DYRCSFXQEWHJB/events.json","paper":"https://pith.science/paper/6QGR4UNH"},"agent_actions":{"view_html":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB","download_json":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB.json","view_paper":"https://pith.science/paper/6QGR4UNH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.20939&json=true","fetch_graph":"https://pith.science/api/pith-number/6QGR4UNH46SL5DYRCSFXQEWHJB/graph.json","fetch_events":"https://pith.science/api/pith-number/6QGR4UNH46SL5DYRCSFXQEWHJB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB/action/storage_attestation","attest_author":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB/action/author_attestation","sign_citation":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB/action/citation_signature","submit_replication":"https://pith.science/pith/6QGR4UNH46SL5DYRCSFXQEWHJB/action/replication_record"}},"created_at":"2026-07-05T11:44:28.604681+00:00","updated_at":"2026-07-05T11:44:28.604681+00:00"}