{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WF25H463T6UTWPMNEKD2W5PLLP","short_pith_number":"pith:WF25H463","schema_version":"1.0","canonical_sha256":"b175d3f3db9fa93b3d8d2287ab75eb5bf01cc5feb4cbe001a5d94a69caf62aaf","source":{"kind":"arxiv","id":"2409.12499","version":2},"attestation_state":"computed","paper":{"title":"End-to-end Open-vocabulary Video Visual Relationship Detection using Multi-modal Prompting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiebo Luo, Shuo Yang, Xinxiao Wu, Yongqi Wang","submitted_at":"2024-09-19T06:25:01Z","abstract_excerpt":"Open-vocabulary video visual relationship detection aims to expand video visual relationship detection beyond annotated categories by detecting unseen relationships between both seen and unseen objects in videos. Existing methods usually use trajectory detectors trained on closed datasets to detect object trajectories, and then feed these trajectories into large-scale pre-trained vision-language models to achieve open-vocabulary classification. Such heavy dependence on the pre-trained trajectory detectors limits their ability to generalize to novel object categories, leading to performance deg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.12499","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-19T06:25:01Z","cross_cats_sorted":[],"title_canon_sha256":"7f4a04bf37df72e5e3beb3df98754c404e4c5c90de265d255b7592ef87546af4","abstract_canon_sha256":"7603d3449a5f526ce72cd5e937daab7d0e377aa8bc4cbd382472ce37c8595262"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:19.462308Z","signature_b64":"Vk5uh+wucmeB2np+L/M+tgaQb3FiaS96E/QDXxAi7SCS+hz/ghr5q9LSYG7UOIb4XpCLMfdUIP3n6A3gN7BmBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b175d3f3db9fa93b3d8d2287ab75eb5bf01cc5feb4cbe001a5d94a69caf62aaf","last_reissued_at":"2026-07-05T10:48:19.461915Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:19.461915Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-end Open-vocabulary Video Visual Relationship Detection using Multi-modal Prompting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiebo Luo, Shuo Yang, Xinxiao Wu, Yongqi Wang","submitted_at":"2024-09-19T06:25:01Z","abstract_excerpt":"Open-vocabulary video visual relationship detection aims to expand video visual relationship detection beyond annotated categories by detecting unseen relationships between both seen and unseen objects in videos. Existing methods usually use trajectory detectors trained on closed datasets to detect object trajectories, and then feed these trajectories into large-scale pre-trained vision-language models to achieve open-vocabulary classification. Such heavy dependence on the pre-trained trajectory detectors limits their ability to generalize to novel object categories, leading to performance deg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.12499","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.12499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.12499","created_at":"2026-07-05T10:48:19.461967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.12499v2","created_at":"2026-07-05T10:48:19.461967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.12499","created_at":"2026-07-05T10:48:19.461967+00:00"},{"alias_kind":"pith_short_12","alias_value":"WF25H463T6UT","created_at":"2026-07-05T10:48:19.461967+00:00"},{"alias_kind":"pith_short_16","alias_value":"WF25H463T6UTWPMN","created_at":"2026-07-05T10:48:19.461967+00:00"},{"alias_kind":"pith_short_8","alias_value":"WF25H463","created_at":"2026-07-05T10:48:19.461967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.17590","citing_title":"DRAMA-X: A Fine-grained Intent Prediction and Risk Reasoning Benchmark For Driving","ref_index":57,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP","json":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP.json","graph_json":"https://pith.science/api/pith-number/WF25H463T6UTWPMNEKD2W5PLLP/graph.json","events_json":"https://pith.science/api/pith-number/WF25H463T6UTWPMNEKD2W5PLLP/events.json","paper":"https://pith.science/paper/WF25H463"},"agent_actions":{"view_html":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP","download_json":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP.json","view_paper":"https://pith.science/paper/WF25H463","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.12499&json=true","fetch_graph":"https://pith.science/api/pith-number/WF25H463T6UTWPMNEKD2W5PLLP/graph.json","fetch_events":"https://pith.science/api/pith-number/WF25H463T6UTWPMNEKD2W5PLLP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP/action/storage_attestation","attest_author":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP/action/author_attestation","sign_citation":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP/action/citation_signature","submit_replication":"https://pith.science/pith/WF25H463T6UTWPMNEKD2W5PLLP/action/replication_record"}},"created_at":"2026-07-05T10:48:19.461967+00:00","updated_at":"2026-07-05T10:48:19.461967+00:00"}