{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OM3Y7QQHTKGFYCLBZNXSQNQTRV","short_pith_number":"pith:OM3Y7QQH","schema_version":"1.0","canonical_sha256":"73378fc2079a8c5c0961cb6f2836138d6664d016fc14a59d3069bdd9e6b7b548","source":{"kind":"arxiv","id":"2504.07519","version":1},"attestation_state":"computed","paper":{"title":"VideoExpert: Augmented LLM for Temporal-Sensitive Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ge-Peng Ji, Henghao Zhao, Huan Xiong, Rui Yan, Zechao Li","submitted_at":"2025-04-10T07:33:39Z","abstract_excerpt":"The core challenge in video understanding lies in perceiving dynamic content changes over time. However, multimodal large language models struggle with temporal-sensitive video tasks, which requires generating timestamps to mark the occurrence of specific events. Existing strategies require MLLMs to generate absolute or relative timestamps directly. We have observed that those MLLMs tend to rely more on language patterns than visual cues when generating timestamps, affecting their performance. To address this problem, we propose VideoExpert, a general-purpose MLLM suitable for several temporal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07519","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-10T07:33:39Z","cross_cats_sorted":[],"title_canon_sha256":"7a7df9e6dd2c9c0be65ef10e3b76946f00cf896db7262044173fb93e834fbd6a","abstract_canon_sha256":"f1b77e6a51b4a62457bf496d33d2d2e289a1dc60878852d0b0eab1dd17d9a77f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:10.283969Z","signature_b64":"hd5Z/tQKsyyDxbjRqIZd9IYr4B83a3BclfRnYeciICucWp209NOZul9xJXSfFSnjGZ3pt+58SeiyQJZHa7wCDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"73378fc2079a8c5c0961cb6f2836138d6664d016fc14a59d3069bdd9e6b7b548","last_reissued_at":"2026-07-05T10:47:10.283542Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:10.283542Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoExpert: Augmented LLM for Temporal-Sensitive Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ge-Peng Ji, Henghao Zhao, Huan Xiong, Rui Yan, Zechao Li","submitted_at":"2025-04-10T07:33:39Z","abstract_excerpt":"The core challenge in video understanding lies in perceiving dynamic content changes over time. However, multimodal large language models struggle with temporal-sensitive video tasks, which requires generating timestamps to mark the occurrence of specific events. Existing strategies require MLLMs to generate absolute or relative timestamps directly. We have observed that those MLLMs tend to rely more on language patterns than visual cues when generating timestamps, affecting their performance. To address this problem, we propose VideoExpert, a general-purpose MLLM suitable for several temporal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07519","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07519/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07519","created_at":"2026-07-05T10:47:10.283598+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07519v1","created_at":"2026-07-05T10:47:10.283598+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07519","created_at":"2026-07-05T10:47:10.283598+00:00"},{"alias_kind":"pith_short_12","alias_value":"OM3Y7QQHTKGF","created_at":"2026-07-05T10:47:10.283598+00:00"},{"alias_kind":"pith_short_16","alias_value":"OM3Y7QQHTKGFYCLB","created_at":"2026-07-05T10:47:10.283598+00:00"},{"alias_kind":"pith_short_8","alias_value":"OM3Y7QQH","created_at":"2026-07-05T10:47:10.283598+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.13377","citing_title":"Time-R1: Post-Training Large Vision Language Model for Temporal Video Grounding","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02860","citing_title":"A Paradigm Shift: Fully End-to-End Training for Temporal Sentence Grounding in Videos","ref_index":78,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV","json":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV.json","graph_json":"https://pith.science/api/pith-number/OM3Y7QQHTKGFYCLBZNXSQNQTRV/graph.json","events_json":"https://pith.science/api/pith-number/OM3Y7QQHTKGFYCLBZNXSQNQTRV/events.json","paper":"https://pith.science/paper/OM3Y7QQH"},"agent_actions":{"view_html":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV","download_json":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV.json","view_paper":"https://pith.science/paper/OM3Y7QQH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07519&json=true","fetch_graph":"https://pith.science/api/pith-number/OM3Y7QQHTKGFYCLBZNXSQNQTRV/graph.json","fetch_events":"https://pith.science/api/pith-number/OM3Y7QQHTKGFYCLBZNXSQNQTRV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV/action/storage_attestation","attest_author":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV/action/author_attestation","sign_citation":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV/action/citation_signature","submit_replication":"https://pith.science/pith/OM3Y7QQHTKGFYCLBZNXSQNQTRV/action/replication_record"}},"created_at":"2026-07-05T10:47:10.283598+00:00","updated_at":"2026-07-05T10:47:10.283598+00:00"}