{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:3V2FUOVGLKAJUJ42Z5XLFPFSYU","short_pith_number":"pith:3V2FUOVG","schema_version":"1.0","canonical_sha256":"dd745a3aa65a809a279acf6eb2bcb2c52256d8db190c38cf1a1eb000c3dc4f8d","source":{"kind":"arxiv","id":"2206.02231","version":3},"attestation_state":"computed","paper":{"title":"Models of human preference for learning reward functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Alessandro Allievi, Peter Stone, Scott Niekum, Serena Booth, Stephane Hatgis-Kessell, W. Bradley Knox","submitted_at":"2022-06-05T17:58:02Z","abstract_excerpt":"The utility of reinforcement learning is limited by the alignment of reward functions with the interests of human stakeholders. One promising method for alignment is to learn the reward function from human-generated preferences between pairs of trajectory segments, a type of reinforcement learning from human feedback (RLHF). These human preferences are typically assumed to be informed solely by partial return, the sum of rewards along each segment. We find this assumption to be flawed and propose modeling human preferences instead as informed by each segment's regret, a measure of a segment's "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.02231","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-05T17:58:02Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"title_canon_sha256":"d82b36090e1b884c71f2701820d7024483dc079f2399abba4172f6175fa0021c","abstract_canon_sha256":"fdb2a99e088a783203be25b87a1227d696195f30168f4dfcf3429c355272874b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:22.609477Z","signature_b64":"P8HWK1vLzhxJgUlIhBduL3S+gb4fU2jVqCSkl4nqo5/gtU68Ct6PiW6GMZAKqv7uRIp/zliL1pZylpOG5tysBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd745a3aa65a809a279acf6eb2bcb2c52256d8db190c38cf1a1eb000c3dc4f8d","last_reissued_at":"2026-07-05T06:48:22.608965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:22.608965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Models of human preference for learning reward functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Alessandro Allievi, Peter Stone, Scott Niekum, Serena Booth, Stephane Hatgis-Kessell, W. Bradley Knox","submitted_at":"2022-06-05T17:58:02Z","abstract_excerpt":"The utility of reinforcement learning is limited by the alignment of reward functions with the interests of human stakeholders. One promising method for alignment is to learn the reward function from human-generated preferences between pairs of trajectory segments, a type of reinforcement learning from human feedback (RLHF). These human preferences are typically assumed to be informed solely by partial return, the sum of rewards along each segment. We find this assumption to be flawed and propose modeling human preferences instead as informed by each segment's regret, a measure of a segment's "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.02231","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.02231/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.02231","created_at":"2026-07-05T06:48:22.609024+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.02231v3","created_at":"2026-07-05T06:48:22.609024+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.02231","created_at":"2026-07-05T06:48:22.609024+00:00"},{"alias_kind":"pith_short_12","alias_value":"3V2FUOVGLKAJ","created_at":"2026-07-05T06:48:22.609024+00:00"},{"alias_kind":"pith_short_16","alias_value":"3V2FUOVGLKAJUJ42","created_at":"2026-07-05T06:48:22.609024+00:00"},{"alias_kind":"pith_short_8","alias_value":"3V2FUOVG","created_at":"2026-07-05T06:48:22.609024+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12603","citing_title":"From Imitation to Alignment: Human-Preference Flow Policies for Long-Horizon Sidewalk Navigation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21822","citing_title":"Implicit Safety Alignment from Crowd Preferences","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06895","citing_title":"Mitigating Cognitive Bias in RLHF by Altering Rationality","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU","json":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU.json","graph_json":"https://pith.science/api/pith-number/3V2FUOVGLKAJUJ42Z5XLFPFSYU/graph.json","events_json":"https://pith.science/api/pith-number/3V2FUOVGLKAJUJ42Z5XLFPFSYU/events.json","paper":"https://pith.science/paper/3V2FUOVG"},"agent_actions":{"view_html":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU","download_json":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU.json","view_paper":"https://pith.science/paper/3V2FUOVG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.02231&json=true","fetch_graph":"https://pith.science/api/pith-number/3V2FUOVGLKAJUJ42Z5XLFPFSYU/graph.json","fetch_events":"https://pith.science/api/pith-number/3V2FUOVGLKAJUJ42Z5XLFPFSYU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/action/storage_attestation","attest_author":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/action/author_attestation","sign_citation":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/action/citation_signature","submit_replication":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/action/replication_record"}},"created_at":"2026-07-05T06:48:22.609024+00:00","updated_at":"2026-07-05T06:48:22.609024+00:00"}