{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FUCHHLD7OSYJ67SWHRAHX2HF6X","short_pith_number":"pith:FUCHHLD7","schema_version":"1.0","canonical_sha256":"2d0473ac7f74b09f7e563c407be8e5f5dd9a8e738398b17ed34dad1418559ad3","source":{"kind":"arxiv","id":"2409.05152","version":2},"attestation_state":"computed","paper":{"title":"OneGen: Efficient One-Pass Unified Generation and Retrieval for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DB","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cheng Peng, Huajun Chen, Jintian Zhang, Jun Zhou, Lei Liang, Mengshu Sun, Ningyu Zhang, Xiang Chen, Zhiqiang Zhang","submitted_at":"2024-09-08T16:35:19Z","abstract_excerpt":"Despite the recent advancements in Large Language Models (LLMs), which have significantly enhanced the generative capabilities for various NLP tasks, LLMs still face limitations in directly handling retrieval tasks. However, many practical applications demand the seamless integration of both retrieval and generation. This paper introduces a novel and efficient One-pass Generation and retrieval framework (OneGen), designed to improve LLMs' performance on tasks that require both generation and retrieval. The proposed framework bridges the traditionally separate training approaches for generation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.05152","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-08T16:35:19Z","cross_cats_sorted":["cs.AI","cs.DB","cs.IR","cs.LG"],"title_canon_sha256":"41a7ccf3f53f88c1b59894aa09dc7bb1cede7cbc8346375759eae715714b9f77","abstract_canon_sha256":"5599de629c04aac8e7437a2fcee6168483fdbff1bf6fb495d120bc19b1650a68"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:38.125293Z","signature_b64":"4t9pKTzEZ0jtVOvOaIN1yNxY5Qz8CVd5Wr7HWJ/ZTpJseCYlW8nh7l5Ac4n8uLFPYOTLdf0tyITre9eL/BEdAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d0473ac7f74b09f7e563c407be8e5f5dd9a8e738398b17ed34dad1418559ad3","last_reissued_at":"2026-07-05T09:14:38.124786Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:38.124786Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OneGen: Efficient One-Pass Unified Generation and Retrieval for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DB","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cheng Peng, Huajun Chen, Jintian Zhang, Jun Zhou, Lei Liang, Mengshu Sun, Ningyu Zhang, Xiang Chen, Zhiqiang Zhang","submitted_at":"2024-09-08T16:35:19Z","abstract_excerpt":"Despite the recent advancements in Large Language Models (LLMs), which have significantly enhanced the generative capabilities for various NLP tasks, LLMs still face limitations in directly handling retrieval tasks. However, many practical applications demand the seamless integration of both retrieval and generation. This paper introduces a novel and efficient One-pass Generation and retrieval framework (OneGen), designed to improve LLMs' performance on tasks that require both generation and retrieval. The proposed framework bridges the traditionally separate training approaches for generation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.05152","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.05152/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.05152","created_at":"2026-07-05T09:14:38.124846+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.05152v2","created_at":"2026-07-05T09:14:38.124846+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.05152","created_at":"2026-07-05T09:14:38.124846+00:00"},{"alias_kind":"pith_short_12","alias_value":"FUCHHLD7OSYJ","created_at":"2026-07-05T09:14:38.124846+00:00"},{"alias_kind":"pith_short_16","alias_value":"FUCHHLD7OSYJ67SW","created_at":"2026-07-05T09:14:38.124846+00:00"},{"alias_kind":"pith_short_8","alias_value":"FUCHHLD7","created_at":"2026-07-05T09:14:38.124846+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23336","citing_title":"Efficient Rationale-based Retrieval: On-policy Distillation from Generative Rerankers based on JEPA","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23336","citing_title":"Efficient Rationale-based Retrieval: On-policy Distillation from Generative Rerankers based on JEPA","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X","json":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X.json","graph_json":"https://pith.science/api/pith-number/FUCHHLD7OSYJ67SWHRAHX2HF6X/graph.json","events_json":"https://pith.science/api/pith-number/FUCHHLD7OSYJ67SWHRAHX2HF6X/events.json","paper":"https://pith.science/paper/FUCHHLD7"},"agent_actions":{"view_html":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X","download_json":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X.json","view_paper":"https://pith.science/paper/FUCHHLD7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.05152&json=true","fetch_graph":"https://pith.science/api/pith-number/FUCHHLD7OSYJ67SWHRAHX2HF6X/graph.json","fetch_events":"https://pith.science/api/pith-number/FUCHHLD7OSYJ67SWHRAHX2HF6X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X/action/storage_attestation","attest_author":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X/action/author_attestation","sign_citation":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X/action/citation_signature","submit_replication":"https://pith.science/pith/FUCHHLD7OSYJ67SWHRAHX2HF6X/action/replication_record"}},"created_at":"2026-07-05T09:14:38.124846+00:00","updated_at":"2026-07-05T09:14:38.124846+00:00"}