{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7WUEXR5AFISRDJUL6EP5NCZR46","short_pith_number":"pith:7WUEXR5A","schema_version":"1.0","canonical_sha256":"fda84bc7a02a2511a68bf11fd68b31e7b507e561ee226af2df634ebfad9c12a9","source":{"kind":"arxiv","id":"2403.10228","version":1},"attestation_state":"computed","paper":{"title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dongyan Zhao, Jianxin Liang, Qun Liu, Xiaojun Meng, Yueqian Wang, Yuxuan Wang","submitted_at":"2024-03-15T11:58:18Z","abstract_excerpt":"Video-text Large Language Models (video-text LLMs) have shown remarkable performance in answering questions and holding conversations on simple videos. However, they perform almost the same as random on grounding text queries in long and complicated videos, having little ability to understand and reason about temporal information, which is the most fundamental difference between videos and images. In this paper, we propose HawkEye, one of the first video-text LLMs that can perform temporal video grounding in a fully text-to-text manner. To collect training data that is applicable for temporal "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.10228","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-15T11:58:18Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"e67022b3d6eddc34c32baeb01fe17bdd2b18495a9854f83f838dc9ceb670c641","abstract_canon_sha256":"0665b8db195606a38e605c2ded51e85de94e1d892d9909407183d0c0e3833b76"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:32.045111Z","signature_b64":"gszb7y6SvEP8W3sLnpkuIPZBm8A5whh6UX/pyadFt5sdmPmhNvyRhQ89NxQpPXziB2UCHJvAU3577HkEjPmJDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fda84bc7a02a2511a68bf11fd68b31e7b507e561ee226af2df634ebfad9c12a9","last_reissued_at":"2026-07-05T07:56:32.044602Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:32.044602Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dongyan Zhao, Jianxin Liang, Qun Liu, Xiaojun Meng, Yueqian Wang, Yuxuan Wang","submitted_at":"2024-03-15T11:58:18Z","abstract_excerpt":"Video-text Large Language Models (video-text LLMs) have shown remarkable performance in answering questions and holding conversations on simple videos. However, they perform almost the same as random on grounding text queries in long and complicated videos, having little ability to understand and reason about temporal information, which is the most fundamental difference between videos and images. In this paper, we propose HawkEye, one of the first video-text LLMs that can perform temporal video grounding in a fully text-to-text manner. To collect training data that is applicable for temporal "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.10228","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.10228/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.10228","created_at":"2026-07-05T07:56:32.044661+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.10228v1","created_at":"2026-07-05T07:56:32.044661+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.10228","created_at":"2026-07-05T07:56:32.044661+00:00"},{"alias_kind":"pith_short_12","alias_value":"7WUEXR5AFISR","created_at":"2026-07-05T07:56:32.044661+00:00"},{"alias_kind":"pith_short_16","alias_value":"7WUEXR5AFISRDJUL","created_at":"2026-07-05T07:56:32.044661+00:00"},{"alias_kind":"pith_short_8","alias_value":"7WUEXR5A","created_at":"2026-07-05T07:56:32.044661+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17798","citing_title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12300","citing_title":"Natural-Language Temporal Grounding in Hour-Long Videos is a Search Problem: A Benchmark and Empirical Decomposition","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09248","citing_title":"Temporal-Aware Reasoning Optimization for Video Temporal Grounding","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06991","citing_title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05758","citing_title":"DRIFT: A Residual Flow Adapter for Decoding Continuous Outputs in Vision-Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25886","citing_title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26104","citing_title":"EVIDENT: Routing MLLM Adaptation through Entity-Grounded Visual Evidence for Cross-Domain Video Temporal Grounding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21973","citing_title":"Foresee-to-Ground: From Predictive Temporal Perception to Evidence-Driven Reasoning for Video Temporal Grounding","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2501.12386","citing_title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03963","citing_title":"TempR1: Improving Temporal Understanding of MLLMs via Temporal-Aware Multi-Task Reinforcement Learning","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06673","citing_title":"Detector-Empowered Video Large Language Model for Efficient Spatio-Temporal Grounding","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13803","citing_title":"EvoGround: Self-Evolving Video Agents for Video Temporal Grounding","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02860","citing_title":"A Paradigm Shift: Fully End-to-End Training for Temporal Sentence Grounding in Videos","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25886","citing_title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25276","citing_title":"OmniVTG: A Large-Scale Dataset and Training Paradigm for Open-World Video Temporal Grounding","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12148","citing_title":"ViLL-E: Video LLM Embeddings for Retrieval","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08966","citing_title":"How Should Video LLMs Output Time? An Analysis of Efficient Temporal Grounding Paradigms","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08522","citing_title":"UniversalVTG: A Universal and Lightweight Foundation Model for Video Temporal Grounding","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08014","citing_title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46","json":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46.json","graph_json":"https://pith.science/api/pith-number/7WUEXR5AFISRDJUL6EP5NCZR46/graph.json","events_json":"https://pith.science/api/pith-number/7WUEXR5AFISRDJUL6EP5NCZR46/events.json","paper":"https://pith.science/paper/7WUEXR5A"},"agent_actions":{"view_html":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46","download_json":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46.json","view_paper":"https://pith.science/paper/7WUEXR5A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.10228&json=true","fetch_graph":"https://pith.science/api/pith-number/7WUEXR5AFISRDJUL6EP5NCZR46/graph.json","fetch_events":"https://pith.science/api/pith-number/7WUEXR5AFISRDJUL6EP5NCZR46/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46/action/storage_attestation","attest_author":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46/action/author_attestation","sign_citation":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46/action/citation_signature","submit_replication":"https://pith.science/pith/7WUEXR5AFISRDJUL6EP5NCZR46/action/replication_record"}},"created_at":"2026-07-05T07:56:32.044661+00:00","updated_at":"2026-07-05T07:56:32.044661+00:00"}