{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:42WOSLCHEIE53WWJFSZOBVWRPR","short_pith_number":"pith:42WOSLCH","schema_version":"1.0","canonical_sha256":"e6ace92c472209dddac92cb2e0d6d17c443ce01f3ba158b37182d31fcffa2a77","source":{"kind":"arxiv","id":"2504.02438","version":5},"attestation_state":"computed","paper":{"title":"Scaling Video-Language Models to 10K Frames via Hierarchical Differential Distillation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chuanqi Cheng, Jian Guan, Rui Yan, Wei Wu","submitted_at":"2025-04-03T09:55:09Z","abstract_excerpt":"Long-form video processing fundamentally challenges vision-language models (VLMs) due to the high computational costs of handling extended temporal sequences. Existing token pruning and feature merging methods often sacrifice critical temporal dependencies or dilute semantic information. We introduce differential distillation, a principled approach that systematically preserves task-relevant information while suppressing redundancy. Based on this principle, we develop ViLAMP, a hierarchical video-language model that processes hour-long videos at \"mixed precision\" through two key mechanisms: (1"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.02438","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-03T09:55:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"80b82f4130d0e37f6654a6b5a5ef64962f5b4221fa66ff762dddb4cf9af109ac","abstract_canon_sha256":"d8b45d0138102f64771703c48375c89aa4409843a2c66976aae2c24cbe50c232"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:08:01.840992Z","signature_b64":"2t/HI9025oKg/cVGoOPoK3n4AoktSRUZFc77kjLvs+17BGxtYmsy0TCEsXW2fGQcxHSr0xylM5FGEsCmI6C4Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6ace92c472209dddac92cb2e0d6d17c443ce01f3ba158b37182d31fcffa2a77","last_reissued_at":"2026-07-05T12:08:01.840497Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:08:01.840497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Video-Language Models to 10K Frames via Hierarchical Differential Distillation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chuanqi Cheng, Jian Guan, Rui Yan, Wei Wu","submitted_at":"2025-04-03T09:55:09Z","abstract_excerpt":"Long-form video processing fundamentally challenges vision-language models (VLMs) due to the high computational costs of handling extended temporal sequences. Existing token pruning and feature merging methods often sacrifice critical temporal dependencies or dilute semantic information. We introduce differential distillation, a principled approach that systematically preserves task-relevant information while suppressing redundancy. Based on this principle, we develop ViLAMP, a hierarchical video-language model that processes hour-long videos at \"mixed precision\" through two key mechanisms: (1"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.02438","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.02438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.02438","created_at":"2026-07-05T12:08:01.840557+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.02438v5","created_at":"2026-07-05T12:08:01.840557+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.02438","created_at":"2026-07-05T12:08:01.840557+00:00"},{"alias_kind":"pith_short_12","alias_value":"42WOSLCHEIE5","created_at":"2026-07-05T12:08:01.840557+00:00"},{"alias_kind":"pith_short_16","alias_value":"42WOSLCHEIE53WWJ","created_at":"2026-07-05T12:08:01.840557+00:00"},{"alias_kind":"pith_short_8","alias_value":"42WOSLCH","created_at":"2026-07-05T12:08:01.840557+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09064","citing_title":"See More, Think Deeper: Query-Expanded Visual Evidence and Answer-Clue Guided Reflection for Long Video Understanding","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06338","citing_title":"StoryVideoQA: Scaling Deep Video Understanding with a Large-Scale, Multi-Genre and Auto-Generated Dataset","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22678","citing_title":"Swift Sampling: Selecting Temporal Surprises via Taylor Series","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06537","citing_title":"MedHorizon: Towards Long-context Medical Video Understanding in the Wild","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR","json":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR.json","graph_json":"https://pith.science/api/pith-number/42WOSLCHEIE53WWJFSZOBVWRPR/graph.json","events_json":"https://pith.science/api/pith-number/42WOSLCHEIE53WWJFSZOBVWRPR/events.json","paper":"https://pith.science/paper/42WOSLCH"},"agent_actions":{"view_html":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR","download_json":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR.json","view_paper":"https://pith.science/paper/42WOSLCH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.02438&json=true","fetch_graph":"https://pith.science/api/pith-number/42WOSLCHEIE53WWJFSZOBVWRPR/graph.json","fetch_events":"https://pith.science/api/pith-number/42WOSLCHEIE53WWJFSZOBVWRPR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR/action/storage_attestation","attest_author":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR/action/author_attestation","sign_citation":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR/action/citation_signature","submit_replication":"https://pith.science/pith/42WOSLCHEIE53WWJFSZOBVWRPR/action/replication_record"}},"created_at":"2026-07-05T12:08:01.840557+00:00","updated_at":"2026-07-05T12:08:01.840557+00:00"}