{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K6I5CDOV3TDYCMYH7JFNJXKLVV","short_pith_number":"pith:K6I5CDOV","schema_version":"1.0","canonical_sha256":"5791d10dd5dcc7813307fa4ad4dd4bad7680281115a8e99e81f1f0b535277dc6","source":{"kind":"arxiv","id":"2503.13983","version":3},"attestation_state":"computed","paper":{"title":"SpaceVLLM: Endowing Multimodal Large Language Model with Spatio-Temporal Video Grounding Capability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongtao Xie, Jiankang Wang, Jiannan Ge, Yang Li, Yongdong Zhang, Zhihang Liu, Zhihan Zhang","submitted_at":"2025-03-18T07:40:36Z","abstract_excerpt":"Multimodal large language models (MLLMs) have made remarkable progress in either temporal or spatial localization. However, they struggle to perform spatio-temporal video grounding. This limitation stems from two major challenges. Firstly, it is difficult to extract accurate spatio-temporal information of each frame in the video. Secondly, the substantial number of visual tokens makes it challenging to precisely map visual tokens of each frame to their corresponding spatial coordinates. To address these issues, we introduce SpaceVLLM, a MLLM endowed with spatio-temporal video grounding capabil"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.13983","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-18T07:40:36Z","cross_cats_sorted":[],"title_canon_sha256":"c963f43f80c463eccbdccb72f4f061d579d48d019cb51bc82c52ac9c3ba09389","abstract_canon_sha256":"032db2d81cf47393e91e227aee8f3f33e62cbd5c68f56f6a667305d23a13abfb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:36.805957Z","signature_b64":"tpOLpoDjxmKru0mH8Iw10nnV7mEEZKhipl7+M/HP9lAjMBH/X6RP/eYltpV7VYSDZ12mpjMMXdyt7gG3+lwbAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5791d10dd5dcc7813307fa4ad4dd4bad7680281115a8e99e81f1f0b535277dc6","last_reissued_at":"2026-07-05T10:47:36.805471Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:36.805471Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpaceVLLM: Endowing Multimodal Large Language Model with Spatio-Temporal Video Grounding Capability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongtao Xie, Jiankang Wang, Jiannan Ge, Yang Li, Yongdong Zhang, Zhihang Liu, Zhihan Zhang","submitted_at":"2025-03-18T07:40:36Z","abstract_excerpt":"Multimodal large language models (MLLMs) have made remarkable progress in either temporal or spatial localization. However, they struggle to perform spatio-temporal video grounding. This limitation stems from two major challenges. Firstly, it is difficult to extract accurate spatio-temporal information of each frame in the video. Secondly, the substantial number of visual tokens makes it challenging to precisely map visual tokens of each frame to their corresponding spatial coordinates. To address these issues, we introduce SpaceVLLM, a MLLM endowed with spatio-temporal video grounding capabil"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13983","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.13983/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.13983","created_at":"2026-07-05T10:47:36.805534+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.13983v3","created_at":"2026-07-05T10:47:36.805534+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13983","created_at":"2026-07-05T10:47:36.805534+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6I5CDOV3TDY","created_at":"2026-07-05T10:47:36.805534+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6I5CDOV3TDYCMYH","created_at":"2026-07-05T10:47:36.805534+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6I5CDOV","created_at":"2026-07-05T10:47:36.805534+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11723","citing_title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13016","citing_title":"SVAG-Bench: A Large-Scale Benchmark for Multi-Instance Spatio-temporal Video Action Grounding","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03666","citing_title":"ToG-Bench: Task-Oriented Spatio-Temporal Grounding in Egocentric Videos","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06673","citing_title":"Detector-Empowered Video Large Language Model for Efficient Spatio-Temporal Grounding","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11723","citing_title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23407","citing_title":"PushupBench: Your VLM is not good at counting pushups","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08014","citing_title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV","json":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV.json","graph_json":"https://pith.science/api/pith-number/K6I5CDOV3TDYCMYH7JFNJXKLVV/graph.json","events_json":"https://pith.science/api/pith-number/K6I5CDOV3TDYCMYH7JFNJXKLVV/events.json","paper":"https://pith.science/paper/K6I5CDOV"},"agent_actions":{"view_html":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV","download_json":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV.json","view_paper":"https://pith.science/paper/K6I5CDOV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.13983&json=true","fetch_graph":"https://pith.science/api/pith-number/K6I5CDOV3TDYCMYH7JFNJXKLVV/graph.json","fetch_events":"https://pith.science/api/pith-number/K6I5CDOV3TDYCMYH7JFNJXKLVV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV/action/storage_attestation","attest_author":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV/action/author_attestation","sign_citation":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV/action/citation_signature","submit_replication":"https://pith.science/pith/K6I5CDOV3TDYCMYH7JFNJXKLVV/action/replication_record"}},"created_at":"2026-07-05T10:47:36.805534+00:00","updated_at":"2026-07-05T10:47:36.805534+00:00"}