{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EMV7OV6GKQOI2SY4TYDYX72AUP","short_pith_number":"pith:EMV7OV6G","schema_version":"1.0","canonical_sha256":"232bf757c6541c8d4b1c9e078bff40a3cccbd156d595feffd95b78193a72aaa0","source":{"kind":"arxiv","id":"2503.10500","version":1},"attestation_state":"computed","paper":{"title":"OmniSTVG: Toward Spatio-Temporal Omni-Object Video Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bing Fan, Heng Fan, Jiali Yao, Libo Zhang, Mengrui Dai, Xin Gu, Xinran Deng, Yan Huang, Zhipeng Zhang","submitted_at":"2025-03-13T16:02:30Z","abstract_excerpt":"In this paper, we propose spatio-temporal omni-object video grounding, dubbed OmniSTVG, a new STVG task that aims at localizing spatially and temporally all targets mentioned in the textual query from videos. Compared to classic STVG locating only a single target, OmniSTVG enables localization of not only an arbitrary number of text-referred targets but also their interacting counterparts in the query from the video, making it more flexible and practical in real scenarios for comprehensive understanding. In order to facilitate exploration of OmniSTVG, we introduce BOSTVG, a large-scale benchma"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10500","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-13T16:02:30Z","cross_cats_sorted":[],"title_canon_sha256":"d6d31ec6eab0734e962dc55c8572a3cb70f3a188fc96a918014373ecb7233a35","abstract_canon_sha256":"e7c6351dee131763140fd0b1b4a83d437839718f9e48f00fd8ecef01f8e86ba4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:46.755806Z","signature_b64":"RJqKE0sMbb6NjvpSSwINrIrr+6r1SFWtd0EQphxFaO3eG5lNiIyYHSfoq26own6sxGhRNv/QO/zUkvki8SyNCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"232bf757c6541c8d4b1c9e078bff40a3cccbd156d595feffd95b78193a72aaa0","last_reissued_at":"2026-07-05T10:30:46.755088Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:46.755088Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OmniSTVG: Toward Spatio-Temporal Omni-Object Video Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bing Fan, Heng Fan, Jiali Yao, Libo Zhang, Mengrui Dai, Xin Gu, Xinran Deng, Yan Huang, Zhipeng Zhang","submitted_at":"2025-03-13T16:02:30Z","abstract_excerpt":"In this paper, we propose spatio-temporal omni-object video grounding, dubbed OmniSTVG, a new STVG task that aims at localizing spatially and temporally all targets mentioned in the textual query from videos. Compared to classic STVG locating only a single target, OmniSTVG enables localization of not only an arbitrary number of text-referred targets but also their interacting counterparts in the query from the video, making it more flexible and practical in real scenarios for comprehensive understanding. In order to facilitate exploration of OmniSTVG, we introduce BOSTVG, a large-scale benchma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10500","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10500/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10500","created_at":"2026-07-05T10:30:46.755180+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10500v1","created_at":"2026-07-05T10:30:46.755180+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10500","created_at":"2026-07-05T10:30:46.755180+00:00"},{"alias_kind":"pith_short_12","alias_value":"EMV7OV6GKQOI","created_at":"2026-07-05T10:30:46.755180+00:00"},{"alias_kind":"pith_short_16","alias_value":"EMV7OV6GKQOI2SY4","created_at":"2026-07-05T10:30:46.755180+00:00"},{"alias_kind":"pith_short_8","alias_value":"EMV7OV6G","created_at":"2026-07-05T10:30:46.755180+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.03666","citing_title":"ToG-Bench: Task-Oriented Spatio-Temporal Grounding in Egocentric Videos","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08014","citing_title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP","json":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP.json","graph_json":"https://pith.science/api/pith-number/EMV7OV6GKQOI2SY4TYDYX72AUP/graph.json","events_json":"https://pith.science/api/pith-number/EMV7OV6GKQOI2SY4TYDYX72AUP/events.json","paper":"https://pith.science/paper/EMV7OV6G"},"agent_actions":{"view_html":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP","download_json":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP.json","view_paper":"https://pith.science/paper/EMV7OV6G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10500&json=true","fetch_graph":"https://pith.science/api/pith-number/EMV7OV6GKQOI2SY4TYDYX72AUP/graph.json","fetch_events":"https://pith.science/api/pith-number/EMV7OV6GKQOI2SY4TYDYX72AUP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP/action/storage_attestation","attest_author":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP/action/author_attestation","sign_citation":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP/action/citation_signature","submit_replication":"https://pith.science/pith/EMV7OV6GKQOI2SY4TYDYX72AUP/action/replication_record"}},"created_at":"2026-07-05T10:30:46.755180+00:00","updated_at":"2026-07-05T10:30:46.755180+00:00"}