{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LMDUVRXQXLAW5B4BZ3AJPIJY2G","short_pith_number":"pith:LMDUVRXQ","schema_version":"1.0","canonical_sha256":"5b074ac6f0bac16e8781cec097a138d197f96f84002207d7098d17e944d7a300","source":{"kind":"arxiv","id":"2305.18438","version":3},"attestation_state":"computed","paper":{"title":"Reinforcement Learning with Human Feedback: Learning Dynamic Choices via Pessimism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Zhuoran Yang, Zihao Li","submitted_at":"2023-05-29T01:18:39Z","abstract_excerpt":"In this paper, we study offline Reinforcement Learning with Human Feedback (RLHF) where we aim to learn the human's underlying reward and the MDP's optimal policy from a set of trajectories induced by human choices. RLHF is challenging for multiple reasons: large state space but limited human feedback, the bounded rationality of human decisions, and the off-policy distribution shift. In this paper, we focus on the Dynamic Discrete Choice (DDC) model for modeling and understanding human choices. DCC, rooted in econometrics and decision theory, is widely used to model a human decision-making pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18438","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-29T01:18:39Z","cross_cats_sorted":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"title_canon_sha256":"1806400cb37180fc2e27c38dfeecbe6e5bd8784395e8748690f1246b9c126060","abstract_canon_sha256":"ce5e0d250fe5eecadb6315fc4b406da60dbecb866dc6beaff91cafb14fa9d5d7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:26:49.810643Z","signature_b64":"7mlPb2CGTVRt7sLXa5UUFhWjoICdjCY/2vQ+ozwmbk+Juo1WAPHM7Vo4mcYw2cK1hBmQ80CI7ABwN3c+OKqvDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b074ac6f0bac16e8781cec097a138d197f96f84002207d7098d17e944d7a300","last_reissued_at":"2026-07-05T06:26:49.810144Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:26:49.810144Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning with Human Feedback: Learning Dynamic Choices via Pessimism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Zhuoran Yang, Zihao Li","submitted_at":"2023-05-29T01:18:39Z","abstract_excerpt":"In this paper, we study offline Reinforcement Learning with Human Feedback (RLHF) where we aim to learn the human's underlying reward and the MDP's optimal policy from a set of trajectories induced by human choices. RLHF is challenging for multiple reasons: large state space but limited human feedback, the bounded rationality of human decisions, and the off-policy distribution shift. In this paper, we focus on the Dynamic Discrete Choice (DDC) model for modeling and understanding human choices. DCC, rooted in econometrics and decision theory, is widely used to model a human decision-making pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18438","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18438","created_at":"2026-07-05T06:26:49.810204+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18438v3","created_at":"2026-07-05T06:26:49.810204+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18438","created_at":"2026-07-05T06:26:49.810204+00:00"},{"alias_kind":"pith_short_12","alias_value":"LMDUVRXQXLAW","created_at":"2026-07-05T06:26:49.810204+00:00"},{"alias_kind":"pith_short_16","alias_value":"LMDUVRXQXLAW5B4B","created_at":"2026-07-05T06:26:49.810204+00:00"},{"alias_kind":"pith_short_8","alias_value":"LMDUVRXQ","created_at":"2026-07-05T06:26:49.810204+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28281","citing_title":"Corruption-robust Offline Multi-agent Reinforcement Learning From Human Feedback","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G","json":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G.json","graph_json":"https://pith.science/api/pith-number/LMDUVRXQXLAW5B4BZ3AJPIJY2G/graph.json","events_json":"https://pith.science/api/pith-number/LMDUVRXQXLAW5B4BZ3AJPIJY2G/events.json","paper":"https://pith.science/paper/LMDUVRXQ"},"agent_actions":{"view_html":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G","download_json":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G.json","view_paper":"https://pith.science/paper/LMDUVRXQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18438&json=true","fetch_graph":"https://pith.science/api/pith-number/LMDUVRXQXLAW5B4BZ3AJPIJY2G/graph.json","fetch_events":"https://pith.science/api/pith-number/LMDUVRXQXLAW5B4BZ3AJPIJY2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/action/storage_attestation","attest_author":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/action/author_attestation","sign_citation":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/action/citation_signature","submit_replication":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/action/replication_record"}},"created_at":"2026-07-05T06:26:49.810204+00:00","updated_at":"2026-07-05T06:26:49.810204+00:00"}