{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FLCHZZMCY2CV37QNDGTM4VKBTK","short_pith_number":"pith:FLCHZZMC","schema_version":"1.0","canonical_sha256":"2ac47ce582c6855dfe0d19a6ce55419aab160e0a02f65740e4352de44d14a575","source":{"kind":"arxiv","id":"2412.12075","version":1},"attestation_state":"computed","paper":{"title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoqi Pei, Guo Chen, Jilan Xu, Limin Wang, Tong Lu, Yali Wang, Yicheng Liu, Yifei Huang, Yuping He","submitted_at":"2024-12-16T18:46:45Z","abstract_excerpt":"Most existing video understanding benchmarks for multimodal large language models (MLLMs) focus only on short videos. The limited number of benchmarks for long video understanding often rely solely on multiple-choice questions (MCQs). However, because of the inherent limitation of MCQ-based evaluation and the increasing reasoning ability of MLLMs, models can give the current answer purely by combining short video understanding with elimination, without genuinely understanding the video content. To address this gap, we introduce CG-Bench, a novel benchmark designed for clue-grounded question an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12075","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-16T18:46:45Z","cross_cats_sorted":[],"title_canon_sha256":"2d350be7e05bb4f5c79f6dcb5fb5eaff5dc475f671bee98510a964c7e12bbcad","abstract_canon_sha256":"c329a8ac4e2d79214acba2aed20e1f2ed48cc870af965d7df330dab49c8d13ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:56.196093Z","signature_b64":"5HstmhwLysH3LkY249uIcyfh9iiYx99bAjU1papNIef/PEUnt6xULKgoenJj7C7Y/IbhI+A4j5s3dbrGrUsiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ac47ce582c6855dfe0d19a6ce55419aab160e0a02f65740e4352de44d14a575","last_reissued_at":"2026-07-05T09:49:56.195647Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:56.195647Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoqi Pei, Guo Chen, Jilan Xu, Limin Wang, Tong Lu, Yali Wang, Yicheng Liu, Yifei Huang, Yuping He","submitted_at":"2024-12-16T18:46:45Z","abstract_excerpt":"Most existing video understanding benchmarks for multimodal large language models (MLLMs) focus only on short videos. The limited number of benchmarks for long video understanding often rely solely on multiple-choice questions (MCQs). However, because of the inherent limitation of MCQ-based evaluation and the increasing reasoning ability of MLLMs, models can give the current answer purely by combining short video understanding with elimination, without genuinely understanding the video content. To address this gap, we introduce CG-Bench, a novel benchmark designed for clue-grounded question an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12075","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12075/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12075","created_at":"2026-07-05T09:49:56.195709+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12075v1","created_at":"2026-07-05T09:49:56.195709+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12075","created_at":"2026-07-05T09:49:56.195709+00:00"},{"alias_kind":"pith_short_12","alias_value":"FLCHZZMCY2CV","created_at":"2026-07-05T09:49:56.195709+00:00"},{"alias_kind":"pith_short_16","alias_value":"FLCHZZMCY2CV37QN","created_at":"2026-07-05T09:49:56.195709+00:00"},{"alias_kind":"pith_short_8","alias_value":"FLCHZZMC","created_at":"2026-07-05T09:49:56.195709+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23216","citing_title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13141","citing_title":"Rethinking RAG in Long Videos: What to Retrieve and How to Use It?","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08231","citing_title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22907","citing_title":"VideoOdyssey: A Benchmark for Ultra-Long-Context and Omni-Modal Video Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23216","citing_title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20715","citing_title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21094","citing_title":"EMCompress: Video-LLMs with Endomorphic Multimodal Compression","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2511.13026","citing_title":"REVISOR: Beyond Textual Reflection, Towards Multimodal Introspective Reasoning in Long-Form Video Understanding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15724","citing_title":"VideoThinker: Building Agentic VideoLLMs with LLM-Guided Tool Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20913","citing_title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09874","citing_title":"EgoMemReason: A Memory-Driven Reasoning Benchmark for Long-Horizon Egocentric Video Understanding","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05546","citing_title":"Efficient Inference for Large Vision-Language Models: Bottlenecks, Techniques, and Prospects","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20937","citing_title":"Sink-Token-Aware Pruning for Fine-Grained Video Understanding in Efficient Video LLMs","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK","json":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK.json","graph_json":"https://pith.science/api/pith-number/FLCHZZMCY2CV37QNDGTM4VKBTK/graph.json","events_json":"https://pith.science/api/pith-number/FLCHZZMCY2CV37QNDGTM4VKBTK/events.json","paper":"https://pith.science/paper/FLCHZZMC"},"agent_actions":{"view_html":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK","download_json":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK.json","view_paper":"https://pith.science/paper/FLCHZZMC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12075&json=true","fetch_graph":"https://pith.science/api/pith-number/FLCHZZMCY2CV37QNDGTM4VKBTK/graph.json","fetch_events":"https://pith.science/api/pith-number/FLCHZZMCY2CV37QNDGTM4VKBTK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK/action/storage_attestation","attest_author":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK/action/author_attestation","sign_citation":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK/action/citation_signature","submit_replication":"https://pith.science/pith/FLCHZZMCY2CV37QNDGTM4VKBTK/action/replication_record"}},"created_at":"2026-07-05T09:49:56.195709+00:00","updated_at":"2026-07-05T09:49:56.195709+00:00"}