{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:VOKQ2PVGRJ7LUGON7BJYZTSJBW","short_pith_number":"pith:VOKQ2PVG","schema_version":"1.0","canonical_sha256":"ab950d3ea68a7eba19cdf8538cce490d8e1ccb8ae28684a338c0ef19553d2826","source":{"kind":"arxiv","id":"2203.07111","version":1},"attestation_state":"computed","paper":{"title":"Disentangled Representation Learning for Text-Video Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Pan Pan, Qiang Wang, Xian-Sheng Hua, Yanhao Zhang, Yun Zheng","submitted_at":"2022-03-14T13:55:33Z","abstract_excerpt":"Cross-modality interaction is a critical component in Text-Video Retrieval (TVR), yet there has been little examination of how different influencing factors for computing interaction affect performance. This paper first studies the interaction paradigm in depth, where we find that its computation can be split into two terms, the interaction contents at different granularity and the matching function to distinguish pairs with the same semantics. We also observe that the single-vector representation and implicit intensive function substantially hinder the optimization. Based on these findings, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.07111","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-03-14T13:55:33Z","cross_cats_sorted":[],"title_canon_sha256":"f3f6e093614aa1dbfe1ebcbbcfd758cd588f921d7f5a40b2b7791fc90472af6b","abstract_canon_sha256":"b497adddbb36b58cf4c8425b24b8aa5cfa2e0e455dde3f8e15ab07631b4a5c30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:59.789980Z","signature_b64":"L/pJNYJkguYhW9GfrMkVs+Cp2EZWYfuroODWn9wCll4+JWnl6ldQ7lOZp1MAFlb/EVgk3PZgw4YiaS1M5tXhDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab950d3ea68a7eba19cdf8538cce490d8e1ccb8ae28684a338c0ef19553d2826","last_reissued_at":"2026-07-05T04:04:59.789558Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:59.789558Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Disentangled Representation Learning for Text-Video Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Pan Pan, Qiang Wang, Xian-Sheng Hua, Yanhao Zhang, Yun Zheng","submitted_at":"2022-03-14T13:55:33Z","abstract_excerpt":"Cross-modality interaction is a critical component in Text-Video Retrieval (TVR), yet there has been little examination of how different influencing factors for computing interaction affect performance. This paper first studies the interaction paradigm in depth, where we find that its computation can be split into two terms, the interaction contents at different granularity and the matching function to distinguish pairs with the same semantics. We also observe that the single-vector representation and implicit intensive function substantially hinder the optimization. Based on these findings, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.07111","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.07111/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.07111","created_at":"2026-07-05T04:04:59.789621+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.07111v1","created_at":"2026-07-05T04:04:59.789621+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.07111","created_at":"2026-07-05T04:04:59.789621+00:00"},{"alias_kind":"pith_short_12","alias_value":"VOKQ2PVGRJ7L","created_at":"2026-07-05T04:04:59.789621+00:00"},{"alias_kind":"pith_short_16","alias_value":"VOKQ2PVGRJ7LUGON","created_at":"2026-07-05T04:04:59.789621+00:00"},{"alias_kind":"pith_short_8","alias_value":"VOKQ2PVG","created_at":"2026-07-05T04:04:59.789621+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00858","citing_title":"MoVA: Learning Asymmetric Dual Projections for Modular Long Video-Text Alignment","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2411.02327","citing_title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17959","citing_title":"Text-Video Retrieval With Global-Local Contrastive Consistency Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06964","citing_title":"Adversarial Video Promotion Against Text-to-Video Retrieval","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2204.00598","citing_title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00826","citing_title":"Understanding the Performance Plateau in Text-to-Video Retrieval: A Comprehensive Empirical and Linguistic Analysis","ref_index":97,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW","json":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW.json","graph_json":"https://pith.science/api/pith-number/VOKQ2PVGRJ7LUGON7BJYZTSJBW/graph.json","events_json":"https://pith.science/api/pith-number/VOKQ2PVGRJ7LUGON7BJYZTSJBW/events.json","paper":"https://pith.science/paper/VOKQ2PVG"},"agent_actions":{"view_html":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW","download_json":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW.json","view_paper":"https://pith.science/paper/VOKQ2PVG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.07111&json=true","fetch_graph":"https://pith.science/api/pith-number/VOKQ2PVGRJ7LUGON7BJYZTSJBW/graph.json","fetch_events":"https://pith.science/api/pith-number/VOKQ2PVGRJ7LUGON7BJYZTSJBW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW/action/storage_attestation","attest_author":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW/action/author_attestation","sign_citation":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW/action/citation_signature","submit_replication":"https://pith.science/pith/VOKQ2PVGRJ7LUGON7BJYZTSJBW/action/replication_record"}},"created_at":"2026-07-05T04:04:59.789621+00:00","updated_at":"2026-07-05T04:04:59.789621+00:00"}