{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JGMMLS6FBO6IUJBVAK7PXIHPDB","short_pith_number":"pith:JGMMLS6F","schema_version":"1.0","canonical_sha256":"4998c5cbc50bbc8a243502befba0ef1857e8895daef302de634fadb93b8173fa","source":{"kind":"arxiv","id":"2412.09907","version":2},"attestation_state":"computed","paper":{"title":"IQViC: In-context, Question Adaptive Vision Compressor for Long-term Video Understanding LMMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Natsuki Miyahara, Shun Takeuchi, Sosuke Yamao, Yuki Harazono","submitted_at":"2024-12-13T06:52:02Z","abstract_excerpt":"With the increasing complexity of video data and the need for more efficient long-term temporal understanding, existing long-term video understanding methods often fail to accurately capture and analyze extended video sequences. These methods typically struggle to maintain performance over longer durations and to handle the intricate dependencies within the video content. To address these limitations, we propose a simple yet effective large multi-modal model framework for long-term video understanding that incorporates a novel visual compressor, the In-context, Question Adaptive Visual Compres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09907","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-13T06:52:02Z","cross_cats_sorted":[],"title_canon_sha256":"220c181fdcfe969cbb71bb2c4bf09e8b5f4c10e47f92d0cbb56c112fda43d883","abstract_canon_sha256":"126ed8ae4e5987da46d065d7f784138e216a897e21dc87330e198885481e1fb3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:14.629907Z","signature_b64":"LReyOnWosAMc6LIpaLsgSy1UmzLdcDBqiPR1rd3ZTMxlUmckyuCJhhSxkda3DrdxbtWqDvFNpdtmHBJsesbRAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4998c5cbc50bbc8a243502befba0ef1857e8895daef302de634fadb93b8173fa","last_reissued_at":"2026-07-05T09:49:14.629442Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:14.629442Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"IQViC: In-context, Question Adaptive Vision Compressor for Long-term Video Understanding LMMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Natsuki Miyahara, Shun Takeuchi, Sosuke Yamao, Yuki Harazono","submitted_at":"2024-12-13T06:52:02Z","abstract_excerpt":"With the increasing complexity of video data and the need for more efficient long-term temporal understanding, existing long-term video understanding methods often fail to accurately capture and analyze extended video sequences. These methods typically struggle to maintain performance over longer durations and to handle the intricate dependencies within the video content. To address these limitations, we propose a simple yet effective large multi-modal model framework for long-term video understanding that incorporates a novel visual compressor, the In-context, Question Adaptive Visual Compres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09907","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09907/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09907","created_at":"2026-07-05T09:49:14.629505+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09907v2","created_at":"2026-07-05T09:49:14.629505+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09907","created_at":"2026-07-05T09:49:14.629505+00:00"},{"alias_kind":"pith_short_12","alias_value":"JGMMLS6FBO6I","created_at":"2026-07-05T09:49:14.629505+00:00"},{"alias_kind":"pith_short_16","alias_value":"JGMMLS6FBO6IUJBV","created_at":"2026-07-05T09:49:14.629505+00:00"},{"alias_kind":"pith_short_8","alias_value":"JGMMLS6F","created_at":"2026-07-05T09:49:14.629505+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB","json":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB.json","graph_json":"https://pith.science/api/pith-number/JGMMLS6FBO6IUJBVAK7PXIHPDB/graph.json","events_json":"https://pith.science/api/pith-number/JGMMLS6FBO6IUJBVAK7PXIHPDB/events.json","paper":"https://pith.science/paper/JGMMLS6F"},"agent_actions":{"view_html":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB","download_json":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB.json","view_paper":"https://pith.science/paper/JGMMLS6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09907&json=true","fetch_graph":"https://pith.science/api/pith-number/JGMMLS6FBO6IUJBVAK7PXIHPDB/graph.json","fetch_events":"https://pith.science/api/pith-number/JGMMLS6FBO6IUJBVAK7PXIHPDB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB/action/storage_attestation","attest_author":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB/action/author_attestation","sign_citation":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB/action/citation_signature","submit_replication":"https://pith.science/pith/JGMMLS6FBO6IUJBVAK7PXIHPDB/action/replication_record"}},"created_at":"2026-07-05T09:49:14.629505+00:00","updated_at":"2026-07-05T09:49:14.629505+00:00"}