{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HTRKOA7DEIEMLUP4Z7FZX57V4T","short_pith_number":"pith:HTRKOA7D","schema_version":"1.0","canonical_sha256":"3ce2a703e32208c5d1fccfcb9bf7f5e4d28f8b9a4ecc70c689b57c1f2db84f08","source":{"kind":"arxiv","id":"2501.05510","version":2},"attestation_state":"computed","paper":{"title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chunjiang Ge, Conghui He, Haodong Duan, Jiaqi Wang, Junbo Niu, Pan Zhang, Qihao He, Rui Qian, Shuangrui Ding, Xiaoyi Dong, Yifei Li, Yuanhang Zhou, Yuhang Cao, Yuhang Zang, Ziyang Miao","submitted_at":"2025-01-09T19:00:01Z","abstract_excerpt":"Temporal Awareness, the ability to reason dynamically based on the timestamp when a question is raised, is the key distinction between offline and online video LLMs. Unlike offline models, which rely on complete videos for static, post hoc analysis, online models process video streams incrementally and dynamically adapt their responses based on the timestamp at which the question is posed. Despite its significance, temporal awareness has not been adequately evaluated in existing benchmarks. To fill this gap, we present OVO-Bench (Online-VideO-Benchmark), a novel video benchmark that emphasizes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05510","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-09T19:00:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1684d4c1eca72b250af01cf3989e323882bfbe6abedd3e354fdcedc80e42590f","abstract_canon_sha256":"f41f3d6e0c2ccfee3eb0cb9385e2c4271718179a5e92e4c08fcbd577eeb3bc03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:16.204241Z","signature_b64":"k69rn6NbqzKjN+DvkHB1lvobwxWgfEvqokB1gYYXWYNw63QU1KraD42uJF29mgmeo8JnM7zabZcenRdV0beADw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ce2a703e32208c5d1fccfcb9bf7f5e4d28f8b9a4ecc70c689b57c1f2db84f08","last_reissued_at":"2026-07-05T10:40:16.203771Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:16.203771Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chunjiang Ge, Conghui He, Haodong Duan, Jiaqi Wang, Junbo Niu, Pan Zhang, Qihao He, Rui Qian, Shuangrui Ding, Xiaoyi Dong, Yifei Li, Yuanhang Zhou, Yuhang Cao, Yuhang Zang, Ziyang Miao","submitted_at":"2025-01-09T19:00:01Z","abstract_excerpt":"Temporal Awareness, the ability to reason dynamically based on the timestamp when a question is raised, is the key distinction between offline and online video LLMs. Unlike offline models, which rely on complete videos for static, post hoc analysis, online models process video streams incrementally and dynamically adapt their responses based on the timestamp at which the question is posed. Despite its significance, temporal awareness has not been adequately evaluated in existing benchmarks. To fill this gap, we present OVO-Bench (Online-VideO-Benchmark), a novel video benchmark that emphasizes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05510","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05510/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05510","created_at":"2026-07-05T10:40:16.203831+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05510v2","created_at":"2026-07-05T10:40:16.203831+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05510","created_at":"2026-07-05T10:40:16.203831+00:00"},{"alias_kind":"pith_short_12","alias_value":"HTRKOA7DEIEM","created_at":"2026-07-05T10:40:16.203831+00:00"},{"alias_kind":"pith_short_16","alias_value":"HTRKOA7DEIEMLUP4","created_at":"2026-07-05T10:40:16.203831+00:00"},{"alias_kind":"pith_short_8","alias_value":"HTRKOA7D","created_at":"2026-07-05T10:40:16.203831+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19849","citing_title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17798","citing_title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01751","citing_title":"MedStreamBench: A Time-Aware Benchmark for Streaming and Proactive Medical Video Understanding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09547","citing_title":"Streaming Interventions: Can Video Large Language Models Correct Mistakes as They Occur?","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06991","citing_title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07639","citing_title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09608","citing_title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21998","citing_title":"Can Multi-Modal LLMs Provide Live Step-by-Step Task Guidance?","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21334","citing_title":"Streaming Video Instruction Tuning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14724","citing_title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22492","citing_title":"MTT-Bench: Predicting Social Dominance in Mice via Multimodal Large Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16893","citing_title":"EasyVideoR1: Easier RL for Video Understanding","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T","json":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T.json","graph_json":"https://pith.science/api/pith-number/HTRKOA7DEIEMLUP4Z7FZX57V4T/graph.json","events_json":"https://pith.science/api/pith-number/HTRKOA7DEIEMLUP4Z7FZX57V4T/events.json","paper":"https://pith.science/paper/HTRKOA7D"},"agent_actions":{"view_html":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T","download_json":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T.json","view_paper":"https://pith.science/paper/HTRKOA7D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05510&json=true","fetch_graph":"https://pith.science/api/pith-number/HTRKOA7DEIEMLUP4Z7FZX57V4T/graph.json","fetch_events":"https://pith.science/api/pith-number/HTRKOA7DEIEMLUP4Z7FZX57V4T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T/action/storage_attestation","attest_author":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T/action/author_attestation","sign_citation":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T/action/citation_signature","submit_replication":"https://pith.science/pith/HTRKOA7DEIEMLUP4Z7FZX57V4T/action/replication_record"}},"created_at":"2026-07-05T10:40:16.203831+00:00","updated_at":"2026-07-05T10:40:16.203831+00:00"}