{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RROPJSG2H5R7X5HM6BDFJBAFMW","short_pith_number":"pith:RROPJSG2","schema_version":"1.0","canonical_sha256":"8c5cf4c8da3f63fbf4ecf04654840565964697dc89b1df6354108857ccfeabdd","source":{"kind":"arxiv","id":"2411.18211","version":1},"attestation_state":"computed","paper":{"title":"TimeMarker: A Versatile Video-LLM for Long and Short Video Understanding with Superior Temporal Localization Ability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Lin Ma, Shimin Chen, Xiaohan Lan, Yitian Yuan, Zequn Jie","submitted_at":"2024-11-27T10:45:40Z","abstract_excerpt":"Rapid development of large language models (LLMs) has significantly advanced multimodal large language models (LMMs), particularly in vision-language tasks. However, existing video-language models often overlook precise temporal localization and struggle with videos of varying lengths. We introduce TimeMarker, a versatile Video-LLM designed for high-quality dialogue based on video content, emphasizing temporal localization. TimeMarker integrates Temporal Separator Tokens to enhance temporal awareness, accurately marking specific moments within videos. It employs the AnyLength mechanism for dyn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.18211","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-27T10:45:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ed1b1c96e6951c5d34802a12337e0a81ffa2ca4da702f240a061e9a9ea00349c","abstract_canon_sha256":"bcfc2a2880a7235a4776279fe837c0d5b464f6042231e760c149df8cda4c4d16"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:13.360133Z","signature_b64":"cHYR4GeEKPEw/Inq3qKArp+VeFk6yJNIPWkPTQ12V2HVtctDZn6gyU2Hv7nzqgCpsUQe5yyvsFNyr/DAkkccAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c5cf4c8da3f63fbf4ecf04654840565964697dc89b1df6354108857ccfeabdd","last_reissued_at":"2026-07-05T09:41:13.359648Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:13.359648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TimeMarker: A Versatile Video-LLM for Long and Short Video Understanding with Superior Temporal Localization Ability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Lin Ma, Shimin Chen, Xiaohan Lan, Yitian Yuan, Zequn Jie","submitted_at":"2024-11-27T10:45:40Z","abstract_excerpt":"Rapid development of large language models (LLMs) has significantly advanced multimodal large language models (LMMs), particularly in vision-language tasks. However, existing video-language models often overlook precise temporal localization and struggle with videos of varying lengths. We introduce TimeMarker, a versatile Video-LLM designed for high-quality dialogue based on video content, emphasizing temporal localization. TimeMarker integrates Temporal Separator Tokens to enhance temporal awareness, accurately marking specific moments within videos. It employs the AnyLength mechanism for dyn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.18211","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.18211/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.18211","created_at":"2026-07-05T09:41:13.359705+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.18211v1","created_at":"2026-07-05T09:41:13.359705+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.18211","created_at":"2026-07-05T09:41:13.359705+00:00"},{"alias_kind":"pith_short_12","alias_value":"RROPJSG2H5R7","created_at":"2026-07-05T09:41:13.359705+00:00"},{"alias_kind":"pith_short_16","alias_value":"RROPJSG2H5R7X5HM","created_at":"2026-07-05T09:41:13.359705+00:00"},{"alias_kind":"pith_short_8","alias_value":"RROPJSG2","created_at":"2026-07-05T09:41:13.359705+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12125","citing_title":"Q-Fold: Query-Aware Focus-Context Spatio-Temporal Folding for Long Video Understanding","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06532","citing_title":"GOPAgen: Motion-Aware and Efficient Agentic Long-Video Understanding with Structural Memory and Hierarchical Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01802","citing_title":"MOSS-Audio Technical Report","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31598","citing_title":"Linear Scaling Video VLMs for Long Video Understanding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18678","citing_title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18678","citing_title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06673","citing_title":"Detector-Empowered Video Large Language Model for Efficient Spatio-Temporal Grounding","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02994","citing_title":"Video-OPD: Efficient Post-Training of Multimodal Large Language Models for Temporal Video Grounding via On-Policy Distillation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20913","citing_title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22267","citing_title":"TiCo: Time-Controllable Spoken Dialogue Model","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07575","citing_title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24893","citing_title":"Interactive Episodic Memory with User Feedback","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08522","citing_title":"UniversalVTG: A Universal and Lightweight Foundation Model for Video Temporal Grounding","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07575","citing_title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13106","citing_title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13023","citing_title":"SpotSound: Enhancing Large Audio-Language Models with Fine-Grained Temporal Grounding","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14149","citing_title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW","json":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW.json","graph_json":"https://pith.science/api/pith-number/RROPJSG2H5R7X5HM6BDFJBAFMW/graph.json","events_json":"https://pith.science/api/pith-number/RROPJSG2H5R7X5HM6BDFJBAFMW/events.json","paper":"https://pith.science/paper/RROPJSG2"},"agent_actions":{"view_html":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW","download_json":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW.json","view_paper":"https://pith.science/paper/RROPJSG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.18211&json=true","fetch_graph":"https://pith.science/api/pith-number/RROPJSG2H5R7X5HM6BDFJBAFMW/graph.json","fetch_events":"https://pith.science/api/pith-number/RROPJSG2H5R7X5HM6BDFJBAFMW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW/action/storage_attestation","attest_author":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW/action/author_attestation","sign_citation":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW/action/citation_signature","submit_replication":"https://pith.science/pith/RROPJSG2H5R7X5HM6BDFJBAFMW/action/replication_record"}},"created_at":"2026-07-05T09:41:13.359705+00:00","updated_at":"2026-07-05T09:41:13.359705+00:00"}