{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OWWV4R7VET3AXRRQ57EF74P4TP","short_pith_number":"pith:OWWV4R7V","schema_version":"1.0","canonical_sha256":"75ad5e47f524f60bc630efc85ff1fc9bea60ac24784b487b8c678046b87048a7","source":{"kind":"arxiv","id":"2404.05083","version":2},"attestation_state":"computed","paper":{"title":"DREAM: Improving Video-Text Retrieval Through Relevance-Based Augmentation Using Large Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo Xue, Mushi Wang, Ning Yu, Shuai Yuan, Wei Pang, Xiangru Jian, Yimu Wang","submitted_at":"2024-04-07T21:46:47Z","abstract_excerpt":"Recent progress in video-text retrieval has been driven largely by advancements in model architectures and training strategies. However, the representation learning capabilities of videotext retrieval models remain constrained by lowquality and limited training data annotations. To address this issue, we present a novel ViDeoText Retrieval Paradigm with RElevance-based AugMentation, namely DREAM, which enhances video and text data using large foundation models to learn more generalized features. Specifically, we first adopt a simple augmentation method, which generates self-similar data by ran"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05083","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-07T21:46:47Z","cross_cats_sorted":["cs.CL","cs.IR","cs.LG"],"title_canon_sha256":"23cd421467445a6079714287637a13ab620eab33df50eea98df352fd01171fa0","abstract_canon_sha256":"6d0916ed43e7c5a87f2c91672aa3af7aec4ffc13996bdb1544d95a809a192f49"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:03.628010Z","signature_b64":"IuHNFlfFlYlWzspjK1jTPzrwpvw1WLlUXoKwy88o3/ReTnpJa52mzMwRCZktOHYG3YMISxLWKZszaB63g1cODA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75ad5e47f524f60bc630efc85ff1fc9bea60ac24784b487b8c678046b87048a7","last_reissued_at":"2026-07-05T10:09:03.627480Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:03.627480Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DREAM: Improving Video-Text Retrieval Through Relevance-Based Augmentation Using Large Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo Xue, Mushi Wang, Ning Yu, Shuai Yuan, Wei Pang, Xiangru Jian, Yimu Wang","submitted_at":"2024-04-07T21:46:47Z","abstract_excerpt":"Recent progress in video-text retrieval has been driven largely by advancements in model architectures and training strategies. However, the representation learning capabilities of videotext retrieval models remain constrained by lowquality and limited training data annotations. To address this issue, we present a novel ViDeoText Retrieval Paradigm with RElevance-based AugMentation, namely DREAM, which enhances video and text data using large foundation models to learn more generalized features. Specifically, we first adopt a simple augmentation method, which generates self-similar data by ran"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05083","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05083/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05083","created_at":"2026-07-05T10:09:03.627544+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05083v2","created_at":"2026-07-05T10:09:03.627544+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05083","created_at":"2026-07-05T10:09:03.627544+00:00"},{"alias_kind":"pith_short_12","alias_value":"OWWV4R7VET3A","created_at":"2026-07-05T10:09:03.627544+00:00"},{"alias_kind":"pith_short_16","alias_value":"OWWV4R7VET3AXRRQ","created_at":"2026-07-05T10:09:03.627544+00:00"},{"alias_kind":"pith_short_8","alias_value":"OWWV4R7V","created_at":"2026-07-05T10:09:03.627544+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.07160","citing_title":"Motion-aware Contrastive Learning for Temporal Panoptic Scene Graph Generation","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP","json":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP.json","graph_json":"https://pith.science/api/pith-number/OWWV4R7VET3AXRRQ57EF74P4TP/graph.json","events_json":"https://pith.science/api/pith-number/OWWV4R7VET3AXRRQ57EF74P4TP/events.json","paper":"https://pith.science/paper/OWWV4R7V"},"agent_actions":{"view_html":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP","download_json":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP.json","view_paper":"https://pith.science/paper/OWWV4R7V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05083&json=true","fetch_graph":"https://pith.science/api/pith-number/OWWV4R7VET3AXRRQ57EF74P4TP/graph.json","fetch_events":"https://pith.science/api/pith-number/OWWV4R7VET3AXRRQ57EF74P4TP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP/action/storage_attestation","attest_author":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP/action/author_attestation","sign_citation":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP/action/citation_signature","submit_replication":"https://pith.science/pith/OWWV4R7VET3AXRRQ57EF74P4TP/action/replication_record"}},"created_at":"2026-07-05T10:09:03.627544+00:00","updated_at":"2026-07-05T10:09:03.627544+00:00"}