{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TCLDYCN5F7AKSRR3ZK7UKQ4OEU","short_pith_number":"pith:TCLDYCN5","schema_version":"1.0","canonical_sha256":"98963c09bd2fc0a9463bcabf45438e252a31ffc5a38ebe252d7eac4071604c8c","source":{"kind":"arxiv","id":"2402.10184","version":7},"attestation_state":"computed","paper":{"title":"Reward Generalization in RLHF: A Topological Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DM"],"primary_cat":"cs.LG","authors_text":"Dong Yan, Fanzhi Zeng, Jiaming Ji, Jiayi Zhou, Josef Dai, Kaile Wang, Tianyi Qiu, Xuehai Pan, Yang Han, Yaodong Yang","submitted_at":"2024-02-15T18:39:24Z","abstract_excerpt":"Existing alignment methods share a common topology of information flow, where reward information is collected from humans, modeled with preference learning, and used to tune language models. However, this shared topology has not been systematically characterized, nor have its alternatives been thoroughly explored, leaving the problems of low data efficiency and unreliable generalization unaddressed. As a solution, we introduce a theory of reward generalization in reinforcement learning from human feedback (RLHF), focusing on the topology of information flow at both macro and micro levels. At t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10184","kind":"arxiv","version":7},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-15T18:39:24Z","cross_cats_sorted":["cs.AI","cs.CL","cs.DM"],"title_canon_sha256":"19fb4c4d047479cd98f386ff62c4b1004714974ccc8e6cd7df0ed6934b872ede","abstract_canon_sha256":"9378df911dbc3a81679bb8ae5f0634919dcc2348af5d20959bef2659853d45b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:45.110841Z","signature_b64":"B1UwBF189DczJfbRzStB+wNKTHai++eCLyTsF0BV1Gp//L7VbiGTaA3cJ8Ta6QVjmie4OUu1bEF0HnhzBFKXBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98963c09bd2fc0a9463bcabf45438e252a31ffc5a38ebe252d7eac4071604c8c","last_reissued_at":"2026-07-05T11:10:45.110330Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:45.110330Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward Generalization in RLHF: A Topological Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DM"],"primary_cat":"cs.LG","authors_text":"Dong Yan, Fanzhi Zeng, Jiaming Ji, Jiayi Zhou, Josef Dai, Kaile Wang, Tianyi Qiu, Xuehai Pan, Yang Han, Yaodong Yang","submitted_at":"2024-02-15T18:39:24Z","abstract_excerpt":"Existing alignment methods share a common topology of information flow, where reward information is collected from humans, modeled with preference learning, and used to tune language models. However, this shared topology has not been systematically characterized, nor have its alternatives been thoroughly explored, leaving the problems of low data efficiency and unreliable generalization unaddressed. As a solution, we introduce a theory of reward generalization in reinforcement learning from human feedback (RLHF), focusing on the topology of information flow at both macro and micro levels. At t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10184","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10184/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10184","created_at":"2026-07-05T11:10:45.110395+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10184v7","created_at":"2026-07-05T11:10:45.110395+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10184","created_at":"2026-07-05T11:10:45.110395+00:00"},{"alias_kind":"pith_short_12","alias_value":"TCLDYCN5F7AK","created_at":"2026-07-05T11:10:45.110395+00:00"},{"alias_kind":"pith_short_16","alias_value":"TCLDYCN5F7AKSRR3","created_at":"2026-07-05T11:10:45.110395+00:00"},{"alias_kind":"pith_short_8","alias_value":"TCLDYCN5","created_at":"2026-07-05T11:10:45.110395+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03489","citing_title":"Learn from Your Mistakes: Tree-like Self-Play for Secure Code LLMs","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU","json":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU.json","graph_json":"https://pith.science/api/pith-number/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/graph.json","events_json":"https://pith.science/api/pith-number/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/events.json","paper":"https://pith.science/paper/TCLDYCN5"},"agent_actions":{"view_html":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU","download_json":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU.json","view_paper":"https://pith.science/paper/TCLDYCN5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10184&json=true","fetch_graph":"https://pith.science/api/pith-number/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/graph.json","fetch_events":"https://pith.science/api/pith-number/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/action/storage_attestation","attest_author":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/action/author_attestation","sign_citation":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/action/citation_signature","submit_replication":"https://pith.science/pith/TCLDYCN5F7AKSRR3ZK7UKQ4OEU/action/replication_record"}},"created_at":"2026-07-05T11:10:45.110395+00:00","updated_at":"2026-07-05T11:10:45.110395+00:00"}