{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:LMDUVRXQXLAW5B4BZ3AJPIJY2G","short_pith_number":"pith:LMDUVRXQ","canonical_record":{"source":{"id":"2305.18438","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-29T01:18:39Z","cross_cats_sorted":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"title_canon_sha256":"1806400cb37180fc2e27c38dfeecbe6e5bd8784395e8748690f1246b9c126060","abstract_canon_sha256":"ce5e0d250fe5eecadb6315fc4b406da60dbecb866dc6beaff91cafb14fa9d5d7"},"schema_version":"1.0"},"canonical_sha256":"5b074ac6f0bac16e8781cec097a138d197f96f84002207d7098d17e944d7a300","source":{"kind":"arxiv","id":"2305.18438","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2305.18438","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"arxiv_version","alias_value":"2305.18438v3","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18438","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"pith_short_12","alias_value":"LMDUVRXQXLAW","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"pith_short_16","alias_value":"LMDUVRXQXLAW5B4B","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"pith_short_8","alias_value":"LMDUVRXQ","created_at":"2026-07-05T06:26:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:LMDUVRXQXLAW5B4BZ3AJPIJY2G","target":"record","payload":{"canonical_record":{"source":{"id":"2305.18438","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-29T01:18:39Z","cross_cats_sorted":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"title_canon_sha256":"1806400cb37180fc2e27c38dfeecbe6e5bd8784395e8748690f1246b9c126060","abstract_canon_sha256":"ce5e0d250fe5eecadb6315fc4b406da60dbecb866dc6beaff91cafb14fa9d5d7"},"schema_version":"1.0"},"canonical_sha256":"5b074ac6f0bac16e8781cec097a138d197f96f84002207d7098d17e944d7a300","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:26:49.810643Z","signature_b64":"7mlPb2CGTVRt7sLXa5UUFhWjoICdjCY/2vQ+ozwmbk+Juo1WAPHM7Vo4mcYw2cK1hBmQ80CI7ABwN3c+OKqvDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b074ac6f0bac16e8781cec097a138d197f96f84002207d7098d17e944d7a300","last_reissued_at":"2026-07-05T06:26:49.810144Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:26:49.810144Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2305.18438","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:26:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rIrGdpBpRE9a5lSeZ3U9pqTjVjEW4xcB9CYZeg2zRWpRFzDDdZ6DlRo6VCo4fMswmJozCsM9aFasglD2DZleCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T01:17:11.068031Z"},"content_sha256":"53edad6088079d9c8ec43453d19000325cfc010f471e71031b1e8e2a2783b77c","schema_version":"1.0","event_id":"sha256:53edad6088079d9c8ec43453d19000325cfc010f471e71031b1e8e2a2783b77c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:LMDUVRXQXLAW5B4BZ3AJPIJY2G","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reinforcement Learning with Human Feedback: Learning Dynamic Choices via Pessimism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Zhuoran Yang, Zihao Li","submitted_at":"2023-05-29T01:18:39Z","abstract_excerpt":"In this paper, we study offline Reinforcement Learning with Human Feedback (RLHF) where we aim to learn the human's underlying reward and the MDP's optimal policy from a set of trajectories induced by human choices. RLHF is challenging for multiple reasons: large state space but limited human feedback, the bounded rationality of human decisions, and the off-policy distribution shift. In this paper, we focus on the Dynamic Discrete Choice (DDC) model for modeling and understanding human choices. DCC, rooted in econometrics and decision theory, is widely used to model a human decision-making pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18438","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:26:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"OxOlikLjONVMKwzIPuU3j0bvdjXXGDehGr1ezVaY92D7Q4DGFFafWzYLoTC47v3u+nPnDzxNLrbPFa1ry6h9Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T01:17:11.068591Z"},"content_sha256":"22505c3dadef45876a6c5e7014767d9bbe9745ddd1d3da358e823f1696a2967f","schema_version":"1.0","event_id":"sha256:22505c3dadef45876a6c5e7014767d9bbe9745ddd1d3da358e823f1696a2967f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/bundle.json","state_url":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T01:17:11Z","links":{"resolver":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G","bundle":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/bundle.json","state":"https://pith.science/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/state.json","well_known_bundle":"https://pith.science/.well-known/pith/LMDUVRXQXLAW5B4BZ3AJPIJY2G/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:LMDUVRXQXLAW5B4BZ3AJPIJY2G","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ce5e0d250fe5eecadb6315fc4b406da60dbecb866dc6beaff91cafb14fa9d5d7","cross_cats_sorted":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-29T01:18:39Z","title_canon_sha256":"1806400cb37180fc2e27c38dfeecbe6e5bd8784395e8748690f1246b9c126060"},"schema_version":"1.0","source":{"id":"2305.18438","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2305.18438","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"arxiv_version","alias_value":"2305.18438v3","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18438","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"pith_short_12","alias_value":"LMDUVRXQXLAW","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"pith_short_16","alias_value":"LMDUVRXQXLAW5B4B","created_at":"2026-07-05T06:26:49Z"},{"alias_kind":"pith_short_8","alias_value":"LMDUVRXQ","created_at":"2026-07-05T06:26:49Z"}],"graph_snapshots":[{"event_id":"sha256:22505c3dadef45876a6c5e7014767d9bbe9745ddd1d3da358e823f1696a2967f","target":"graph","created_at":"2026-07-05T06:26:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2305.18438/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this paper, we study offline Reinforcement Learning with Human Feedback (RLHF) where we aim to learn the human's underlying reward and the MDP's optimal policy from a set of trajectories induced by human choices. RLHF is challenging for multiple reasons: large state space but limited human feedback, the bounded rationality of human decisions, and the off-policy distribution shift. In this paper, we focus on the Dynamic Discrete Choice (DDC) model for modeling and understanding human choices. DCC, rooted in econometrics and decision theory, is widely used to model a human decision-making pro","authors_text":"Mengdi Wang, Zhuoran Yang, Zihao Li","cross_cats":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-29T01:18:39Z","title":"Reinforcement Learning with Human Feedback: Learning Dynamic Choices via Pessimism"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18438","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:53edad6088079d9c8ec43453d19000325cfc010f471e71031b1e8e2a2783b77c","target":"record","created_at":"2026-07-05T06:26:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ce5e0d250fe5eecadb6315fc4b406da60dbecb866dc6beaff91cafb14fa9d5d7","cross_cats_sorted":["cs.AI","math.OC","math.ST","stat.ML","stat.TH"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-29T01:18:39Z","title_canon_sha256":"1806400cb37180fc2e27c38dfeecbe6e5bd8784395e8748690f1246b9c126060"},"schema_version":"1.0","source":{"id":"2305.18438","kind":"arxiv","version":3}},"canonical_sha256":"5b074ac6f0bac16e8781cec097a138d197f96f84002207d7098d17e944d7a300","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5b074ac6f0bac16e8781cec097a138d197f96f84002207d7098d17e944d7a300","first_computed_at":"2026-07-05T06:26:49.810144Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:26:49.810144Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"7mlPb2CGTVRt7sLXa5UUFhWjoICdjCY/2vQ+ozwmbk+Juo1WAPHM7Vo4mcYw2cK1hBmQ80CI7ABwN3c+OKqvDg==","signature_status":"signed_v1","signed_at":"2026-07-05T06:26:49.810643Z","signed_message":"canonical_sha256_bytes"},"source_id":"2305.18438","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:53edad6088079d9c8ec43453d19000325cfc010f471e71031b1e8e2a2783b77c","sha256:22505c3dadef45876a6c5e7014767d9bbe9745ddd1d3da358e823f1696a2967f"],"state_sha256":"e4243e21fe3560a3c0ce2fadf36fddab0e66fa5677a02d0359332edd16cbc715"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5jjR8QmpZmfFfieBzBG43F/GdgOwh1zZEC11NWy3Jc0Imd/NrjgV6yj7T9pJzPgqy2xhjuyvcaBzLpc1x/XSBw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T01:17:11.072351Z","bundle_sha256":"176831113872120f6747d55d59cfeb00928b2a6dd89badb9bfd0e585f4b4f272"}}