{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D4HIWLQQEYH4CGSE24PWPSTS53","short_pith_number":"pith:D4HIWLQQ","schema_version":"1.0","canonical_sha256":"1f0e8b2e10260fc11a44d71f67ca72eed1e9f1c230793156f039e041cb460d36","source":{"kind":"arxiv","id":"2412.17415","version":2},"attestation_state":"computed","paper":{"title":"VidCtx: Context-aware Video Question Answering with Image Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Andreas Goulas, Ioannis Patras, Vasileios Mezaris","submitted_at":"2024-12-23T09:26:38Z","abstract_excerpt":"To address computational and memory limitations of Large Multimodal Models in the Video Question-Answering task, several recent methods extract textual representations per frame (e.g., by captioning) and feed them to a Large Language Model (LLM) that processes them to produce the final response. However, in this way, the LLM does not have access to visual information and often has to process repetitive textual descriptions of nearby frames. To address those shortcomings, in this paper, we introduce VidCtx, a novel training-free VideoQA framework which integrates both modalities, i.e. both visu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.17415","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-23T09:26:38Z","cross_cats_sorted":["cs.AI","cs.MM"],"title_canon_sha256":"1cc2812599e9bc3615b9e26e8773f533214f92830aed177e485252dd8ca1879b","abstract_canon_sha256":"a2dac37d500b0f68ee645fcd3e5d10736294d3be4a6e52afc201cf1ddbf205a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:15.579172Z","signature_b64":"9S+KdluBaqDFoV94gfjZtb56wsaGfWH/Sq7x76yp+cT82xuGoXtOTBQF1qbpS9CAlD6traVWJ3EOjixKwkt5CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f0e8b2e10260fc11a44d71f67ca72eed1e9f1c230793156f039e041cb460d36","last_reissued_at":"2026-07-05T10:45:15.578724Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:15.578724Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VidCtx: Context-aware Video Question Answering with Image Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Andreas Goulas, Ioannis Patras, Vasileios Mezaris","submitted_at":"2024-12-23T09:26:38Z","abstract_excerpt":"To address computational and memory limitations of Large Multimodal Models in the Video Question-Answering task, several recent methods extract textual representations per frame (e.g., by captioning) and feed them to a Large Language Model (LLM) that processes them to produce the final response. However, in this way, the LLM does not have access to visual information and often has to process repetitive textual descriptions of nearby frames. To address those shortcomings, in this paper, we introduce VidCtx, a novel training-free VideoQA framework which integrates both modalities, i.e. both visu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.17415","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.17415/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.17415","created_at":"2026-07-05T10:45:15.578787+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.17415v2","created_at":"2026-07-05T10:45:15.578787+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.17415","created_at":"2026-07-05T10:45:15.578787+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4HIWLQQEYH4","created_at":"2026-07-05T10:45:15.578787+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4HIWLQQEYH4CGSE","created_at":"2026-07-05T10:45:15.578787+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4HIWLQQ","created_at":"2026-07-05T10:45:15.578787+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05736","citing_title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53","json":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53.json","graph_json":"https://pith.science/api/pith-number/D4HIWLQQEYH4CGSE24PWPSTS53/graph.json","events_json":"https://pith.science/api/pith-number/D4HIWLQQEYH4CGSE24PWPSTS53/events.json","paper":"https://pith.science/paper/D4HIWLQQ"},"agent_actions":{"view_html":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53","download_json":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53.json","view_paper":"https://pith.science/paper/D4HIWLQQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.17415&json=true","fetch_graph":"https://pith.science/api/pith-number/D4HIWLQQEYH4CGSE24PWPSTS53/graph.json","fetch_events":"https://pith.science/api/pith-number/D4HIWLQQEYH4CGSE24PWPSTS53/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53/action/storage_attestation","attest_author":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53/action/author_attestation","sign_citation":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53/action/citation_signature","submit_replication":"https://pith.science/pith/D4HIWLQQEYH4CGSE24PWPSTS53/action/replication_record"}},"created_at":"2026-07-05T10:45:15.578787+00:00","updated_at":"2026-07-05T10:45:15.578787+00:00"}