{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TY5U7XPGTGRLSFEWU6TZHK7SJF","short_pith_number":"pith:TY5U7XPG","schema_version":"1.0","canonical_sha256":"9e3b4fdde699a2b91496a7a793abf249590757b38a107744ddaa31b6ede5c3ca","source":{"kind":"arxiv","id":"2403.19046","version":1},"attestation_state":"computed","paper":{"title":"LITA: Language Instructed Temporal-Localization Assistant","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"De-An Huang, Hongxu Yin, Jan Kautz, Pavlo Molchanov, Shijia Liao, Subhashree Radhakrishnan, Zhiding Yu","submitted_at":"2024-03-27T22:50:48Z","abstract_excerpt":"There has been tremendous progress in multimodal Large Language Models (LLMs). Recent works have extended these models to video input with promising instruction following capabilities. However, an important missing piece is temporal localization. These models cannot accurately answer the \"When?\" questions. We identify three key aspects that limit their temporal localization capabilities: (i) time representation, (ii) architecture, and (iii) data. We address these shortcomings by proposing Language Instructed Temporal-Localization Assistant (LITA) with the following features: (1) We introduce t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.19046","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-27T22:50:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d8ffc341938dd0f08b94efda8ec9f602d585bfcef1b4391f53dc3013d8d88980","abstract_canon_sha256":"f758cd960eaaa283369123344bfd17af227d86b88b7b6e87d16c743e48e6b6b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:01:43.442421Z","signature_b64":"gF4Qi808v/qcl2WSPzdM5+C/kGRIZjIscVRM2yNyYc0GD4YO5H8rp1XM5a3jx8+BuARedL/iN+t+ikuE5ZDqDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e3b4fdde699a2b91496a7a793abf249590757b38a107744ddaa31b6ede5c3ca","last_reissued_at":"2026-07-05T08:01:43.441971Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:01:43.441971Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LITA: Language Instructed Temporal-Localization Assistant","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"De-An Huang, Hongxu Yin, Jan Kautz, Pavlo Molchanov, Shijia Liao, Subhashree Radhakrishnan, Zhiding Yu","submitted_at":"2024-03-27T22:50:48Z","abstract_excerpt":"There has been tremendous progress in multimodal Large Language Models (LLMs). Recent works have extended these models to video input with promising instruction following capabilities. However, an important missing piece is temporal localization. These models cannot accurately answer the \"When?\" questions. We identify three key aspects that limit their temporal localization capabilities: (i) time representation, (ii) architecture, and (iii) data. We address these shortcomings by proposing Language Instructed Temporal-Localization Assistant (LITA) with the following features: (1) We introduce t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19046","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.19046/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.19046","created_at":"2026-07-05T08:01:43.442030+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.19046v1","created_at":"2026-07-05T08:01:43.442030+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19046","created_at":"2026-07-05T08:01:43.442030+00:00"},{"alias_kind":"pith_short_12","alias_value":"TY5U7XPGTGRL","created_at":"2026-07-05T08:01:43.442030+00:00"},{"alias_kind":"pith_short_16","alias_value":"TY5U7XPGTGRLSFEW","created_at":"2026-07-05T08:01:43.442030+00:00"},{"alias_kind":"pith_short_8","alias_value":"TY5U7XPG","created_at":"2026-07-05T08:01:43.442030+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.25886","citing_title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16994","citing_title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13803","citing_title":"EvoGround: Self-Evolving Video Agents for Video Temporal Grounding","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02860","citing_title":"A Paradigm Shift: Fully End-to-End Training for Temporal Sentence Grounding in Videos","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25886","citing_title":"MarkIt: Training-Free Visual Markers for Precise Video Temporal Grounding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02713","citing_title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","ref_index":188,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF","json":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF.json","graph_json":"https://pith.science/api/pith-number/TY5U7XPGTGRLSFEWU6TZHK7SJF/graph.json","events_json":"https://pith.science/api/pith-number/TY5U7XPGTGRLSFEWU6TZHK7SJF/events.json","paper":"https://pith.science/paper/TY5U7XPG"},"agent_actions":{"view_html":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF","download_json":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF.json","view_paper":"https://pith.science/paper/TY5U7XPG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.19046&json=true","fetch_graph":"https://pith.science/api/pith-number/TY5U7XPGTGRLSFEWU6TZHK7SJF/graph.json","fetch_events":"https://pith.science/api/pith-number/TY5U7XPGTGRLSFEWU6TZHK7SJF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF/action/storage_attestation","attest_author":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF/action/author_attestation","sign_citation":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF/action/citation_signature","submit_replication":"https://pith.science/pith/TY5U7XPGTGRLSFEWU6TZHK7SJF/action/replication_record"}},"created_at":"2026-07-05T08:01:43.442030+00:00","updated_at":"2026-07-05T08:01:43.442030+00:00"}