{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PCXZU4M4NX5KW2NAMQRE3YCCJ4","short_pith_number":"pith:PCXZU4M4","schema_version":"1.0","canonical_sha256":"78af9a719c6dfaab69a064224de0424f08a2a05ad43926e398e7a5a57fd3f409","source":{"kind":"arxiv","id":"2503.09416","version":1},"attestation_state":"computed","paper":{"title":"OpenVidVRD: Open-Vocabulary Video Visual Relation Detection via Prompt-Driven Semantic Space Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Qi Liu, Weiying Xue, Yuxiao Wang, Zhenao Wei","submitted_at":"2025-03-12T14:13:17Z","abstract_excerpt":"The video visual relation detection (VidVRD) task is to identify objects and their relationships in videos, which is challenging due to the dynamic content, high annotation costs, and long-tailed distribution of relations. Visual language models (VLMs) help explore open-vocabulary visual relation detection tasks, yet often overlook the connections between various visual regions and their relations. Moreover, using VLMs to directly identify visual relations in videos poses significant challenges because of the large disparity between images and videos. Therefore, we propose a novel open-vocabul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.09416","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-12T14:13:17Z","cross_cats_sorted":[],"title_canon_sha256":"a23cd14af11bf9d7c9902cb148c891285f77df61803e2134039ef1010bb5bdbb","abstract_canon_sha256":"ff972f431917aab21a6874a73113f5b81f4c45f79e09703b53addabfa44f67e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:29:55.891218Z","signature_b64":"e+d2fEDlUnSjdBo7GQ86B+m0lB0jnBEEcO2czyKDDcrC1s+8W/pND7K5Bj24Hddgl1pQ9EdxW7VyVuXebQwJCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78af9a719c6dfaab69a064224de0424f08a2a05ad43926e398e7a5a57fd3f409","last_reissued_at":"2026-07-05T10:29:55.890707Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:29:55.890707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenVidVRD: Open-Vocabulary Video Visual Relation Detection via Prompt-Driven Semantic Space Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Qi Liu, Weiying Xue, Yuxiao Wang, Zhenao Wei","submitted_at":"2025-03-12T14:13:17Z","abstract_excerpt":"The video visual relation detection (VidVRD) task is to identify objects and their relationships in videos, which is challenging due to the dynamic content, high annotation costs, and long-tailed distribution of relations. Visual language models (VLMs) help explore open-vocabulary visual relation detection tasks, yet often overlook the connections between various visual regions and their relations. Moreover, using VLMs to directly identify visual relations in videos poses significant challenges because of the large disparity between images and videos. Therefore, we propose a novel open-vocabul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.09416","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.09416/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.09416","created_at":"2026-07-05T10:29:55.890779+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.09416v1","created_at":"2026-07-05T10:29:55.890779+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.09416","created_at":"2026-07-05T10:29:55.890779+00:00"},{"alias_kind":"pith_short_12","alias_value":"PCXZU4M4NX5K","created_at":"2026-07-05T10:29:55.890779+00:00"},{"alias_kind":"pith_short_16","alias_value":"PCXZU4M4NX5KW2NA","created_at":"2026-07-05T10:29:55.890779+00:00"},{"alias_kind":"pith_short_8","alias_value":"PCXZU4M4","created_at":"2026-07-05T10:29:55.890779+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4","json":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4.json","graph_json":"https://pith.science/api/pith-number/PCXZU4M4NX5KW2NAMQRE3YCCJ4/graph.json","events_json":"https://pith.science/api/pith-number/PCXZU4M4NX5KW2NAMQRE3YCCJ4/events.json","paper":"https://pith.science/paper/PCXZU4M4"},"agent_actions":{"view_html":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4","download_json":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4.json","view_paper":"https://pith.science/paper/PCXZU4M4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.09416&json=true","fetch_graph":"https://pith.science/api/pith-number/PCXZU4M4NX5KW2NAMQRE3YCCJ4/graph.json","fetch_events":"https://pith.science/api/pith-number/PCXZU4M4NX5KW2NAMQRE3YCCJ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4/action/storage_attestation","attest_author":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4/action/author_attestation","sign_citation":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4/action/citation_signature","submit_replication":"https://pith.science/pith/PCXZU4M4NX5KW2NAMQRE3YCCJ4/action/replication_record"}},"created_at":"2026-07-05T10:29:55.890779+00:00","updated_at":"2026-07-05T10:29:55.890779+00:00"}