{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3OADC3LLRV5IVHQHZYVUDOFCNN","short_pith_number":"pith:3OADC3LL","schema_version":"1.0","canonical_sha256":"db80316d6b8d7a8a9e07ce2b41b8a26b50f70cb98981427634a9823bff668e8e","source":{"kind":"arxiv","id":"2303.00001","version":1},"attestation_state":"computed","paper":{"title":"Reward Design with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Dorsa Sadigh, Kalesha Bullard, Minae Kwon, Sang Michael Xie","submitted_at":"2023-02-27T22:09:35Z","abstract_excerpt":"Reward design in reinforcement learning (RL) is challenging since specifying human notions of desired behavior may be difficult via reward functions or require many expert demonstrations. Can we instead cheaply design rewards using a natural language interface? This paper explores how to simplify reward design by prompting a large language model (LLM) such as GPT-3 as a proxy reward function, where the user provides a textual prompt containing a few examples (few-shot) or a description (zero-shot) of the desired behavior. Our approach leverages this proxy reward function in an RL framework. Sp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.00001","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-27T22:09:35Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"874d4a3eccabe2573d69ea329ca57b4c2bf816cc519545d72109a07cec4fe318","abstract_canon_sha256":"1c3bbe2f7549b011bc3d641d32ad5eaa7169a8cbb0ea4f5f2ef568c695489427"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:46:53.538869Z","signature_b64":"JmRPgo2SFNiUAYHqaKljOH8UUEICOhWC7dNzS/xPYkiUZCaIjYyctSbApcinDJGSOHkbo6ceIRv/olv8qFRYDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db80316d6b8d7a8a9e07ce2b41b8a26b50f70cb98981427634a9823bff668e8e","last_reissued_at":"2026-07-05T05:46:53.538431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:46:53.538431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward Design with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Dorsa Sadigh, Kalesha Bullard, Minae Kwon, Sang Michael Xie","submitted_at":"2023-02-27T22:09:35Z","abstract_excerpt":"Reward design in reinforcement learning (RL) is challenging since specifying human notions of desired behavior may be difficult via reward functions or require many expert demonstrations. Can we instead cheaply design rewards using a natural language interface? This paper explores how to simplify reward design by prompting a large language model (LLM) such as GPT-3 as a proxy reward function, where the user provides a textual prompt containing a few examples (few-shot) or a description (zero-shot) of the desired behavior. Our approach leverages this proxy reward function in an RL framework. Sp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.00001","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.00001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.00001","created_at":"2026-07-05T05:46:53.538496+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.00001v1","created_at":"2026-07-05T05:46:53.538496+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.00001","created_at":"2026-07-05T05:46:53.538496+00:00"},{"alias_kind":"pith_short_12","alias_value":"3OADC3LLRV5I","created_at":"2026-07-05T05:46:53.538496+00:00"},{"alias_kind":"pith_short_16","alias_value":"3OADC3LLRV5IVHQH","created_at":"2026-07-05T05:46:53.538496+00:00"},{"alias_kind":"pith_short_8","alias_value":"3OADC3LL","created_at":"2026-07-05T05:46:53.538496+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10528","citing_title":"Representation-Aware Advantage Estimation: Your Reward Model Provides More Than A Scalar Output","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06673","citing_title":"Uncertainty-Aware LLM-Guided Policy Shaping for Sparse-Reward Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29869","citing_title":"ARKD: Adaptive Reinforcement Learning-Guided Bidirectional KL Divergence Distillation for Text Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2512.00778","citing_title":"What Is Preference Optimization Doing, and Why?","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17900","citing_title":"DuIVRS-2: An LLM-based Interactive Voice Response System for Large-scale POI Attribute Acquisition","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16615","citing_title":"LLM-Guided Task- and Affordance-Level Exploration in Reinforcement Learning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2307.05973","citing_title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11859","citing_title":"EvoNav: Evolutionary Reward Function Design for Robot Navigation with Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08315","citing_title":"Reflective Prompted Policy Optimization: Trajectory-Grounded Revision and Salience Bias","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01123","citing_title":"PERSA: Reinforcement Learning for Professor-Style Personalized Feedback with LLMs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10517","citing_title":"From Perception to Planning: Evolving Ego-Centric Task-Oriented Spatiotemporal Reasoning via Curriculum Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02073","citing_title":"Enhanced LLM Reasoning by Optimizing Reward Functions with Search-Driven Reinforcement Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02073","citing_title":"Enhanced LLM Reasoning by Optimizing Reward Functions with Search-Driven Reinforcement Learning","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN","json":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN.json","graph_json":"https://pith.science/api/pith-number/3OADC3LLRV5IVHQHZYVUDOFCNN/graph.json","events_json":"https://pith.science/api/pith-number/3OADC3LLRV5IVHQHZYVUDOFCNN/events.json","paper":"https://pith.science/paper/3OADC3LL"},"agent_actions":{"view_html":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN","download_json":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN.json","view_paper":"https://pith.science/paper/3OADC3LL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.00001&json=true","fetch_graph":"https://pith.science/api/pith-number/3OADC3LLRV5IVHQHZYVUDOFCNN/graph.json","fetch_events":"https://pith.science/api/pith-number/3OADC3LLRV5IVHQHZYVUDOFCNN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN/action/storage_attestation","attest_author":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN/action/author_attestation","sign_citation":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN/action/citation_signature","submit_replication":"https://pith.science/pith/3OADC3LLRV5IVHQHZYVUDOFCNN/action/replication_record"}},"created_at":"2026-07-05T05:46:53.538496+00:00","updated_at":"2026-07-05T05:46:53.538496+00:00"}