{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ESHO6CD2BMD4FKPNNCAAA73KH5","short_pith_number":"pith:ESHO6CD2","schema_version":"1.0","canonical_sha256":"248eef087a0b07c2a9ed6880007f6a3f4fee7c691c1edc7e99eb7f7a4925463a","source":{"kind":"arxiv","id":"2306.11816","version":2},"attestation_state":"computed","paper":{"title":"Learning to Generate Better Than Your LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Dipendra Misra, Jonathan D. Chang, Kiante Brantley, Rajkumar Ramamurthy, Wen Sun","submitted_at":"2023-06-20T18:19:17Z","abstract_excerpt":"Reinforcement learning (RL) has emerged as a powerful paradigm for fine-tuning Large Language Models (LLMs) for text generation. In particular, recent LLMs such as ChatGPT and GPT-4 can engage in fluent conversations with users after finetuning with RL. Capitalizing on key properties of text generation, we seek to investigate RL algorithms beyond general purpose algorithms like Proximal Policy Optimization (PPO). In particular, we extend RL algorithms to allow them to interact with a dynamic black-box guide LLM and propose RL with guided feedback (RLGF), a suite of RL algorithms for LLM fine-t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.11816","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-20T18:19:17Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ca1bd6edbc0b90ef31f40269722041661cada8e6c2bb7a8d2500482a21433524","abstract_canon_sha256":"1e3592f618604604e93436034a1315169791a64055404cff09083c786d5882dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:56.713435Z","signature_b64":"ddlxnV80GARKJWaX2ptke+xZ7qhoaGDmooAGp1QgGEO8ZrWr/w1oVFXMInR+4gDTWeWhsm/ZutDnYYzlyYKjAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"248eef087a0b07c2a9ed6880007f6a3f4fee7c691c1edc7e99eb7f7a4925463a","last_reissued_at":"2026-07-05T07:11:56.713058Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:56.713058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Generate Better Than Your LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Dipendra Misra, Jonathan D. Chang, Kiante Brantley, Rajkumar Ramamurthy, Wen Sun","submitted_at":"2023-06-20T18:19:17Z","abstract_excerpt":"Reinforcement learning (RL) has emerged as a powerful paradigm for fine-tuning Large Language Models (LLMs) for text generation. In particular, recent LLMs such as ChatGPT and GPT-4 can engage in fluent conversations with users after finetuning with RL. Capitalizing on key properties of text generation, we seek to investigate RL algorithms beyond general purpose algorithms like Proximal Policy Optimization (PPO). In particular, we extend RL algorithms to allow them to interact with a dynamic black-box guide LLM and propose RL with guided feedback (RLGF), a suite of RL algorithms for LLM fine-t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.11816","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.11816/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.11816","created_at":"2026-07-05T07:11:56.713115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.11816v2","created_at":"2026-07-05T07:11:56.713115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.11816","created_at":"2026-07-05T07:11:56.713115+00:00"},{"alias_kind":"pith_short_12","alias_value":"ESHO6CD2BMD4","created_at":"2026-07-05T07:11:56.713115+00:00"},{"alias_kind":"pith_short_16","alias_value":"ESHO6CD2BMD4FKPN","created_at":"2026-07-05T07:11:56.713115+00:00"},{"alias_kind":"pith_short_8","alias_value":"ESHO6CD2","created_at":"2026-07-05T07:11:56.713115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.08146","citing_title":"Rewarding Progress: Scaling Automated Process Verifiers for LLM Reasoning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5","json":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5.json","graph_json":"https://pith.science/api/pith-number/ESHO6CD2BMD4FKPNNCAAA73KH5/graph.json","events_json":"https://pith.science/api/pith-number/ESHO6CD2BMD4FKPNNCAAA73KH5/events.json","paper":"https://pith.science/paper/ESHO6CD2"},"agent_actions":{"view_html":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5","download_json":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5.json","view_paper":"https://pith.science/paper/ESHO6CD2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.11816&json=true","fetch_graph":"https://pith.science/api/pith-number/ESHO6CD2BMD4FKPNNCAAA73KH5/graph.json","fetch_events":"https://pith.science/api/pith-number/ESHO6CD2BMD4FKPNNCAAA73KH5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5/action/storage_attestation","attest_author":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5/action/author_attestation","sign_citation":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5/action/citation_signature","submit_replication":"https://pith.science/pith/ESHO6CD2BMD4FKPNNCAAA73KH5/action/replication_record"}},"created_at":"2026-07-05T07:11:56.713115+00:00","updated_at":"2026-07-05T07:11:56.713115+00:00"}