{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:X662LZJJMJRRO64MS6ZUH5N7DJ","short_pith_number":"pith:X662LZJJ","schema_version":"1.0","canonical_sha256":"bfbda5e5296263177b8c97b343f5bf1a72fe0380194698b01a234049fac1ee85","source":{"kind":"arxiv","id":"2312.08358","version":2},"attestation_state":"computed","paper":{"title":"Distributional Preference Learning: Understanding and Accounting for Hidden Context in RLHF","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anand Siththaranjan, Cassidy Laidlaw, Dylan Hadfield-Menell","submitted_at":"2023-12-13T18:51:34Z","abstract_excerpt":"In practice, preference learning from human feedback depends on incomplete data with hidden context. Hidden context refers to data that affects the feedback received, but which is not represented in the data used to train a preference model. This captures common issues of data collection, such as having human annotators with varied preferences, cognitive processes that result in seemingly irrational behavior, and combining data labeled according to different criteria. We prove that standard applications of preference learning, including reinforcement learning from human feedback (RLHF), implic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.08358","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-13T18:51:34Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"bb9ac2a0754658d3cb15384c0e6126ccf0bd3007448cdf9d2067c9933985f412","abstract_canon_sha256":"7296a150875687fc2d60c97bfa66a0b7389c37a60fef877a216198942b3c54bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:53.476691Z","signature_b64":"TAjj1xvl6fRj2sKFMO+j2O9EweCyxKM+v4Fb8daTK8PuQiasyrEzfHcpfLvA4RuAq1djMx1v0XES1ypGD+tODg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bfbda5e5296263177b8c97b343f5bf1a72fe0380194698b01a234049fac1ee85","last_reissued_at":"2026-07-05T08:08:53.476266Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:53.476266Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distributional Preference Learning: Understanding and Accounting for Hidden Context in RLHF","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anand Siththaranjan, Cassidy Laidlaw, Dylan Hadfield-Menell","submitted_at":"2023-12-13T18:51:34Z","abstract_excerpt":"In practice, preference learning from human feedback depends on incomplete data with hidden context. Hidden context refers to data that affects the feedback received, but which is not represented in the data used to train a preference model. This captures common issues of data collection, such as having human annotators with varied preferences, cognitive processes that result in seemingly irrational behavior, and combining data labeled according to different criteria. We prove that standard applications of preference learning, including reinforcement learning from human feedback (RLHF), implic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.08358","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.08358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.08358","created_at":"2026-07-05T08:08:53.476328+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.08358v2","created_at":"2026-07-05T08:08:53.476328+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.08358","created_at":"2026-07-05T08:08:53.476328+00:00"},{"alias_kind":"pith_short_12","alias_value":"X662LZJJMJRR","created_at":"2026-07-05T08:08:53.476328+00:00"},{"alias_kind":"pith_short_16","alias_value":"X662LZJJMJRRO64M","created_at":"2026-07-05T08:08:53.476328+00:00"},{"alias_kind":"pith_short_8","alias_value":"X662LZJJ","created_at":"2026-07-05T08:08:53.476328+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":215,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09038","citing_title":"Personalization Meets Safety:Mechanisms,Risks,and Mitigations in Personalized LLMs","ref_index":186,"is_internal_anchor":false},{"citing_arxiv_id":"2412.08812","citing_title":"Test-Time Alignment via Hypothesis Reweighting","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03238","citing_title":"RLHF May Not Reflect Genuine Preferences","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09876","citing_title":"Efficient Personalization of Generative User Interfaces","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20805","citing_title":"Relative Principals, Pluralistic Alignment, and the Structural Value Alignment Problem","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ","json":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ.json","graph_json":"https://pith.science/api/pith-number/X662LZJJMJRRO64MS6ZUH5N7DJ/graph.json","events_json":"https://pith.science/api/pith-number/X662LZJJMJRRO64MS6ZUH5N7DJ/events.json","paper":"https://pith.science/paper/X662LZJJ"},"agent_actions":{"view_html":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ","download_json":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ.json","view_paper":"https://pith.science/paper/X662LZJJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.08358&json=true","fetch_graph":"https://pith.science/api/pith-number/X662LZJJMJRRO64MS6ZUH5N7DJ/graph.json","fetch_events":"https://pith.science/api/pith-number/X662LZJJMJRRO64MS6ZUH5N7DJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ/action/storage_attestation","attest_author":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ/action/author_attestation","sign_citation":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ/action/citation_signature","submit_replication":"https://pith.science/pith/X662LZJJMJRRO64MS6ZUH5N7DJ/action/replication_record"}},"created_at":"2026-07-05T08:08:53.476328+00:00","updated_at":"2026-07-05T08:08:53.476328+00:00"}