{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YNC2YE3ZYZ6KPKRTHNG23XAVLO","short_pith_number":"pith:YNC2YE3Z","schema_version":"1.0","canonical_sha256":"c345ac1379c67ca7aa333b4daddc155b8ed06c4a5b86d41105009043e8c0feb7","source":{"kind":"arxiv","id":"2502.03824","version":3},"attestation_state":"computed","paper":{"title":"Syntriever: How to Train Your Retriever with Synthetic Data from LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Minsang Kim, Seungjun Baek","submitted_at":"2025-02-06T07:19:59Z","abstract_excerpt":"LLMs have boosted progress in many AI applications. Recently, there were attempts to distill the vast knowledge of LLMs into information retrieval systems. Those distillation methods mostly use output probabilities of LLMs which are unavailable in the latest black-box LLMs. We propose Syntriever, a training framework for retrievers using synthetic data from black-box LLMs. Syntriever consists of two stages. Firstly in the distillation stage, we synthesize relevant and plausibly irrelevant passages and augmented queries using chain-of-thoughts for the given queries. LLM is asked to self-verify "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03824","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-06T07:19:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d486d437925cc5af7d09aa1f104f561e7cce377ddd0e436e0503c426184c9341","abstract_canon_sha256":"c25b81b4195d7ed5cfc4151b08956cd733cdfc81ba0fa5e8c4eb00e376bd9995"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:08.493131Z","signature_b64":"3wC+uARb0ZrrndcNVoz1xgqr4iqaJ6NrXb41RBl+H0ZZ4oj2I7og/nOhmcwdXrim7zkyFbPS4iSPe7ZaBqYVDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c345ac1379c67ca7aa333b4daddc155b8ed06c4a5b86d41105009043e8c0feb7","last_reissued_at":"2026-07-05T10:14:08.492642Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:08.492642Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Syntriever: How to Train Your Retriever with Synthetic Data from LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Minsang Kim, Seungjun Baek","submitted_at":"2025-02-06T07:19:59Z","abstract_excerpt":"LLMs have boosted progress in many AI applications. Recently, there were attempts to distill the vast knowledge of LLMs into information retrieval systems. Those distillation methods mostly use output probabilities of LLMs which are unavailable in the latest black-box LLMs. We propose Syntriever, a training framework for retrievers using synthetic data from black-box LLMs. Syntriever consists of two stages. Firstly in the distillation stage, we synthesize relevant and plausibly irrelevant passages and augmented queries using chain-of-thoughts for the given queries. LLM is asked to self-verify "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03824","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03824/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03824","created_at":"2026-07-05T10:14:08.492700+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03824v3","created_at":"2026-07-05T10:14:08.492700+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03824","created_at":"2026-07-05T10:14:08.492700+00:00"},{"alias_kind":"pith_short_12","alias_value":"YNC2YE3ZYZ6K","created_at":"2026-07-05T10:14:08.492700+00:00"},{"alias_kind":"pith_short_16","alias_value":"YNC2YE3ZYZ6KPKRT","created_at":"2026-07-05T10:14:08.492700+00:00"},{"alias_kind":"pith_short_8","alias_value":"YNC2YE3Z","created_at":"2026-07-05T10:14:08.492700+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12744","citing_title":"GRIP: Feedback-Guided Prompt Retrieval for Large Multimodal Models","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO","json":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO.json","graph_json":"https://pith.science/api/pith-number/YNC2YE3ZYZ6KPKRTHNG23XAVLO/graph.json","events_json":"https://pith.science/api/pith-number/YNC2YE3ZYZ6KPKRTHNG23XAVLO/events.json","paper":"https://pith.science/paper/YNC2YE3Z"},"agent_actions":{"view_html":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO","download_json":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO.json","view_paper":"https://pith.science/paper/YNC2YE3Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03824&json=true","fetch_graph":"https://pith.science/api/pith-number/YNC2YE3ZYZ6KPKRTHNG23XAVLO/graph.json","fetch_events":"https://pith.science/api/pith-number/YNC2YE3ZYZ6KPKRTHNG23XAVLO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO/action/storage_attestation","attest_author":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO/action/author_attestation","sign_citation":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO/action/citation_signature","submit_replication":"https://pith.science/pith/YNC2YE3ZYZ6KPKRTHNG23XAVLO/action/replication_record"}},"created_at":"2026-07-05T10:14:08.492700+00:00","updated_at":"2026-07-05T10:14:08.492700+00:00"}