{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:5AVENWBWJLVIO7D3FG4E7TLYE2","short_pith_number":"pith:5AVENWBW","canonical_record":{"source":{"id":"2503.13538","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-15T20:53:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6e81d88aebd138c2bc16a0612cd1f2d497fdd9dca322a77e05cedbe11dac0331","abstract_canon_sha256":"ded4a35c4e72373f69c5746c3446f19866d664c06d8c08f485005753545fb995"},"schema_version":"1.0"},"canonical_sha256":"e82a46d8364aea877c7b29b84fcd78269f8dc10845fc72c7ca40cb691429eafa","source":{"kind":"arxiv","id":"2503.13538","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2503.13538","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"arxiv_version","alias_value":"2503.13538v1","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13538","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"pith_short_12","alias_value":"5AVENWBWJLVI","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"pith_short_16","alias_value":"5AVENWBWJLVIO7D3","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"pith_short_8","alias_value":"5AVENWBW","created_at":"2026-07-05T10:33:12Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:5AVENWBWJLVIO7D3FG4E7TLYE2","target":"record","payload":{"canonical_record":{"source":{"id":"2503.13538","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-15T20:53:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6e81d88aebd138c2bc16a0612cd1f2d497fdd9dca322a77e05cedbe11dac0331","abstract_canon_sha256":"ded4a35c4e72373f69c5746c3446f19866d664c06d8c08f485005753545fb995"},"schema_version":"1.0"},"canonical_sha256":"e82a46d8364aea877c7b29b84fcd78269f8dc10845fc72c7ca40cb691429eafa","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:33:12.335061Z","signature_b64":"RJFwbAOVg19j1tNGsT7N2uPDAp8abD8ypxOjSJRxLb5mFPsJ9Cx4g3H+Y85DjEOtgVaPjmo8iTslFf3ywOM0DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e82a46d8364aea877c7b29b84fcd78269f8dc10845fc72c7ca40cb691429eafa","last_reissued_at":"2026-07-05T10:33:12.334427Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:33:12.334427Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2503.13538","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:33:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5lVpT3JsnjLtVNa5bRge9xlbMednBqevP0XQTETYf53npgYM2c4/+dgjWv38ecjWfd4MXaI4oT16t0HtiziQCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T19:55:10.398283Z"},"content_sha256":"31ec8c741bac6003448bf2ea656408c362f9049cd8c5ff6a47d12d4b35b7352c","schema_version":"1.0","event_id":"sha256:31ec8c741bac6003448bf2ea656408c362f9049cd8c5ff6a47d12d4b35b7352c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:5AVENWBWJLVIO7D3FG4E7TLYE2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"From Demonstrations to Rewards: Alignment Without Explicit Human Preferences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"George Karypis, Huzefa Rangwala, Mingyi Hong, Rasool Fakoor, Siliang Zeng, Yao Liu","submitted_at":"2025-03-15T20:53:46Z","abstract_excerpt":"One of the challenges of aligning large models with human preferences lies in both the data requirements and the technical complexities of current approaches. Predominant methods, such as RLHF, involve multiple steps, each demanding distinct types of data, including demonstration data and preference data. In RLHF, human preferences are typically modeled through a reward model, which serves as a proxy to guide policy learning during the reinforcement learning stage, ultimately producing a policy aligned with human preferences. However, in this paper, we propose a fresh perspective on learning a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13538","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.13538/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:33:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"LrGf3C1rrXqc17NsE0GPqgL5PXR25E9KVqOC55tsiRVyjtE06TALTg2gmAsPaVZspAgB4s1GJkmRDg+ynuSWBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T19:55:10.399329Z"},"content_sha256":"8051adbd46565e23a1f582786ef86296a912bad205877fb7593f5789a29d6da1","schema_version":"1.0","event_id":"sha256:8051adbd46565e23a1f582786ef86296a912bad205877fb7593f5789a29d6da1"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/5AVENWBWJLVIO7D3FG4E7TLYE2/bundle.json","state_url":"https://pith.science/pith/5AVENWBWJLVIO7D3FG4E7TLYE2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/5AVENWBWJLVIO7D3FG4E7TLYE2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T19:55:10Z","links":{"resolver":"https://pith.science/pith/5AVENWBWJLVIO7D3FG4E7TLYE2","bundle":"https://pith.science/pith/5AVENWBWJLVIO7D3FG4E7TLYE2/bundle.json","state":"https://pith.science/pith/5AVENWBWJLVIO7D3FG4E7TLYE2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/5AVENWBWJLVIO7D3FG4E7TLYE2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:5AVENWBWJLVIO7D3FG4E7TLYE2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ded4a35c4e72373f69c5746c3446f19866d664c06d8c08f485005753545fb995","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-15T20:53:46Z","title_canon_sha256":"6e81d88aebd138c2bc16a0612cd1f2d497fdd9dca322a77e05cedbe11dac0331"},"schema_version":"1.0","source":{"id":"2503.13538","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2503.13538","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"arxiv_version","alias_value":"2503.13538v1","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13538","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"pith_short_12","alias_value":"5AVENWBWJLVI","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"pith_short_16","alias_value":"5AVENWBWJLVIO7D3","created_at":"2026-07-05T10:33:12Z"},{"alias_kind":"pith_short_8","alias_value":"5AVENWBW","created_at":"2026-07-05T10:33:12Z"}],"graph_snapshots":[{"event_id":"sha256:8051adbd46565e23a1f582786ef86296a912bad205877fb7593f5789a29d6da1","target":"graph","created_at":"2026-07-05T10:33:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2503.13538/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"One of the challenges of aligning large models with human preferences lies in both the data requirements and the technical complexities of current approaches. Predominant methods, such as RLHF, involve multiple steps, each demanding distinct types of data, including demonstration data and preference data. In RLHF, human preferences are typically modeled through a reward model, which serves as a proxy to guide policy learning during the reinforcement learning stage, ultimately producing a policy aligned with human preferences. However, in this paper, we propose a fresh perspective on learning a","authors_text":"George Karypis, Huzefa Rangwala, Mingyi Hong, Rasool Fakoor, Siliang Zeng, Yao Liu","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-15T20:53:46Z","title":"From Demonstrations to Rewards: Alignment Without Explicit Human Preferences"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13538","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:31ec8c741bac6003448bf2ea656408c362f9049cd8c5ff6a47d12d4b35b7352c","target":"record","created_at":"2026-07-05T10:33:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ded4a35c4e72373f69c5746c3446f19866d664c06d8c08f485005753545fb995","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-15T20:53:46Z","title_canon_sha256":"6e81d88aebd138c2bc16a0612cd1f2d497fdd9dca322a77e05cedbe11dac0331"},"schema_version":"1.0","source":{"id":"2503.13538","kind":"arxiv","version":1}},"canonical_sha256":"e82a46d8364aea877c7b29b84fcd78269f8dc10845fc72c7ca40cb691429eafa","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e82a46d8364aea877c7b29b84fcd78269f8dc10845fc72c7ca40cb691429eafa","first_computed_at":"2026-07-05T10:33:12.334427Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:33:12.334427Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"RJFwbAOVg19j1tNGsT7N2uPDAp8abD8ypxOjSJRxLb5mFPsJ9Cx4g3H+Y85DjEOtgVaPjmo8iTslFf3ywOM0DQ==","signature_status":"signed_v1","signed_at":"2026-07-05T10:33:12.335061Z","signed_message":"canonical_sha256_bytes"},"source_id":"2503.13538","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:31ec8c741bac6003448bf2ea656408c362f9049cd8c5ff6a47d12d4b35b7352c","sha256:8051adbd46565e23a1f582786ef86296a912bad205877fb7593f5789a29d6da1"],"state_sha256":"925c23ea623ad4783c0be5bf500828d46ce8963e84dbd1905311a99951c33aa9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ah6x7BoytJnckuhzruLsW9VbRQtTeBLApJ5NB5CKlrX6DpFlZasD0mpYRrbJMRkUtcc44zxbkA54O5NbH2ftAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T19:55:10.405214Z","bundle_sha256":"56329f9e9904a0c051e4d87b39aee2df0f3650858036d65d073b8d0fefd7d13d"}}