{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VMOVOUA6H43OKULK4GFXITVJ5X","short_pith_number":"pith:VMOVOUA6","schema_version":"1.0","canonical_sha256":"ab1d57501e3f36e5516ae18b744ea9edee4a8313627c5c6c81a43725c303a686","source":{"kind":"arxiv","id":"2307.07184","version":3},"attestation_state":"computed","paper":{"title":"TVPR: Text-to-Video Person Retrieval and a New Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aichun Zhu, Fan Ni, Guan-Nan Dong, Hui Liu, Jianhui Wu, Mingcheng Ni, Xu Zhang","submitted_at":"2023-07-14T06:34:00Z","abstract_excerpt":"Most existing methods for text-based person retrieval focus on text-to-image person retrieval. Nevertheless, due to the lack of dynamic information provided by isolated frames, the performance is hampered when the person is obscured or variable motion details are missed in isolated frames. To overcome this, we propose a novel Text-to-Video Person Retrieval (TVPR) task. Since there is no dataset or benchmark that describes person videos with natural language, we construct a large-scale cross-modal person video dataset containing detailed natural language annotations, termed as Text-to-Video Per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.07184","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-07-14T06:34:00Z","cross_cats_sorted":[],"title_canon_sha256":"025846aca2e321eb5bd8f41ff90cce2c48d811bfb488e0b58e1e5db000f6ef5d","abstract_canon_sha256":"082a251fe6cfa37a1e8a5077de4e730e306b7a2016a0f52e9c843e07fad98fa4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:18.238292Z","signature_b64":"Z29K7nKsw7ZRuCoqMT6yB6r45h20/luze/qafszyTgRCD+V6vA4+c3T2qYAVdIg8xqw/5TEvkPBQ7Vr8O3EsAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab1d57501e3f36e5516ae18b744ea9edee4a8313627c5c6c81a43725c303a686","last_reissued_at":"2026-07-05T10:51:18.237814Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:18.237814Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TVPR: Text-to-Video Person Retrieval and a New Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aichun Zhu, Fan Ni, Guan-Nan Dong, Hui Liu, Jianhui Wu, Mingcheng Ni, Xu Zhang","submitted_at":"2023-07-14T06:34:00Z","abstract_excerpt":"Most existing methods for text-based person retrieval focus on text-to-image person retrieval. Nevertheless, due to the lack of dynamic information provided by isolated frames, the performance is hampered when the person is obscured or variable motion details are missed in isolated frames. To overcome this, we propose a novel Text-to-Video Person Retrieval (TVPR) task. Since there is no dataset or benchmark that describes person videos with natural language, we construct a large-scale cross-modal person video dataset containing detailed natural language annotations, termed as Text-to-Video Per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.07184","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.07184/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.07184","created_at":"2026-07-05T10:51:18.237871+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.07184v3","created_at":"2026-07-05T10:51:18.237871+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.07184","created_at":"2026-07-05T10:51:18.237871+00:00"},{"alias_kind":"pith_short_12","alias_value":"VMOVOUA6H43O","created_at":"2026-07-05T10:51:18.237871+00:00"},{"alias_kind":"pith_short_16","alias_value":"VMOVOUA6H43OKULK","created_at":"2026-07-05T10:51:18.237871+00:00"},{"alias_kind":"pith_short_8","alias_value":"VMOVOUA6","created_at":"2026-07-05T10:51:18.237871+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X","json":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X.json","graph_json":"https://pith.science/api/pith-number/VMOVOUA6H43OKULK4GFXITVJ5X/graph.json","events_json":"https://pith.science/api/pith-number/VMOVOUA6H43OKULK4GFXITVJ5X/events.json","paper":"https://pith.science/paper/VMOVOUA6"},"agent_actions":{"view_html":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X","download_json":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X.json","view_paper":"https://pith.science/paper/VMOVOUA6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.07184&json=true","fetch_graph":"https://pith.science/api/pith-number/VMOVOUA6H43OKULK4GFXITVJ5X/graph.json","fetch_events":"https://pith.science/api/pith-number/VMOVOUA6H43OKULK4GFXITVJ5X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X/action/storage_attestation","attest_author":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X/action/author_attestation","sign_citation":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X/action/citation_signature","submit_replication":"https://pith.science/pith/VMOVOUA6H43OKULK4GFXITVJ5X/action/replication_record"}},"created_at":"2026-07-05T10:51:18.237871+00:00","updated_at":"2026-07-05T10:51:18.237871+00:00"}