{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:3QNL2SYJW6SEBIE5LGDNW634JN","short_pith_number":"pith:3QNL2SYJ","canonical_record":{"source":{"id":"2509.22851","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-26T19:03:24Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"04cd33ead47ebe9916bae403c5a8daad273bf65ce2e58f7f4db244cd94382956","abstract_canon_sha256":"07acff6c39f6c9cfa8cef061901f7c7e7c036c48208c6b27b0cc7a08b134c120"},"schema_version":"1.0"},"canonical_sha256":"dc1abd4b09b7a440a09d5986db7b7c4b7d98cc5b71506f6de777c5fa6802ba15","source":{"kind":"arxiv","id":"2509.22851","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.22851","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"arxiv_version","alias_value":"2509.22851v4","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.22851","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"pith_short_12","alias_value":"3QNL2SYJW6SE","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"pith_short_16","alias_value":"3QNL2SYJW6SEBIE5","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"pith_short_8","alias_value":"3QNL2SYJ","created_at":"2026-07-07T00:15:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:3QNL2SYJW6SEBIE5LGDNW634JN","target":"record","payload":{"canonical_record":{"source":{"id":"2509.22851","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-26T19:03:24Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"04cd33ead47ebe9916bae403c5a8daad273bf65ce2e58f7f4db244cd94382956","abstract_canon_sha256":"07acff6c39f6c9cfa8cef061901f7c7e7c036c48208c6b27b0cc7a08b134c120"},"schema_version":"1.0"},"canonical_sha256":"dc1abd4b09b7a440a09d5986db7b7c4b7d98cc5b71506f6de777c5fa6802ba15","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T00:15:49.939906Z","signature_b64":"yqXLL7tP+CdLVhG0UNkslqigDT7s7ey6MfLpNw5fWcLnS1rosaI7SyXWxkPMXDXDo9lrDYzZitGW8m8ppM45CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc1abd4b09b7a440a09d5986db7b7c4b7d98cc5b71506f6de777c5fa6802ba15","last_reissued_at":"2026-07-07T00:15:49.939184Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T00:15:49.939184Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2509.22851","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T00:15:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"i9Gz1WqFkj610LVMbcJEiZQC90Ibi887JPsp9qSkUMqzcH8jkIaITRS+3lFn8s5P92+VWbCI4uqw2KueRIp4CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T01:30:35.464713Z"},"content_sha256":"64c3fc5763f1a24ad4e083df19fbc5856074217ffcbfe5bc90d648c84a1359d3","schema_version":"1.0","event_id":"sha256:64c3fc5763f1a24ad4e083df19fbc5856074217ffcbfe5bc90d648c84a1359d3"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:3QNL2SYJW6SEBIE5LGDNW634JN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Adaptive Margin RLHF via Preference over Preferences","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Greg Durrett, Prasann Singhal, Scott Niekum, Yaswanth Chittepu","submitted_at":"2025-09-26T19:03:24Z","abstract_excerpt":"Margin-based optimization is fundamental to improving generalization and robustness in classification tasks. In the context of reward model learning from preferences within Reinforcement Learning from Human Feedback (RLHF), existing methods typically rely on no margins, fixed margins, or margins that are simplistic functions of preference ratings. However, such formulations often fail to account for the varying strengths of different preferences or they rely on noisy margin information derived from preference ratings. Furthermore, many existing methods that use adaptive margins assume access t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.22851","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.22851/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T00:15:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"C550am3HZRQfnqwuVqAjoUj7oOHxZmycgaE2A3s+HCoaDpnxD119Yjm6F0vt24L3dsetWR+bQK3u9SAo632SBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T01:30:35.465643Z"},"content_sha256":"a98ae3e53243afb0adbd235c485ca216ada71be2b0adac99e4d337c289cd0f8e","schema_version":"1.0","event_id":"sha256:a98ae3e53243afb0adbd235c485ca216ada71be2b0adac99e4d337c289cd0f8e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3QNL2SYJW6SEBIE5LGDNW634JN/bundle.json","state_url":"https://pith.science/pith/3QNL2SYJW6SEBIE5LGDNW634JN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3QNL2SYJW6SEBIE5LGDNW634JN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T01:30:35Z","links":{"resolver":"https://pith.science/pith/3QNL2SYJW6SEBIE5LGDNW634JN","bundle":"https://pith.science/pith/3QNL2SYJW6SEBIE5LGDNW634JN/bundle.json","state":"https://pith.science/pith/3QNL2SYJW6SEBIE5LGDNW634JN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3QNL2SYJW6SEBIE5LGDNW634JN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:3QNL2SYJW6SEBIE5LGDNW634JN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"07acff6c39f6c9cfa8cef061901f7c7e7c036c48208c6b27b0cc7a08b134c120","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-26T19:03:24Z","title_canon_sha256":"04cd33ead47ebe9916bae403c5a8daad273bf65ce2e58f7f4db244cd94382956"},"schema_version":"1.0","source":{"id":"2509.22851","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.22851","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"arxiv_version","alias_value":"2509.22851v4","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.22851","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"pith_short_12","alias_value":"3QNL2SYJW6SE","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"pith_short_16","alias_value":"3QNL2SYJW6SEBIE5","created_at":"2026-07-07T00:15:49Z"},{"alias_kind":"pith_short_8","alias_value":"3QNL2SYJ","created_at":"2026-07-07T00:15:49Z"}],"graph_snapshots":[{"event_id":"sha256:a98ae3e53243afb0adbd235c485ca216ada71be2b0adac99e4d337c289cd0f8e","target":"graph","created_at":"2026-07-07T00:15:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2509.22851/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Margin-based optimization is fundamental to improving generalization and robustness in classification tasks. In the context of reward model learning from preferences within Reinforcement Learning from Human Feedback (RLHF), existing methods typically rely on no margins, fixed margins, or margins that are simplistic functions of preference ratings. However, such formulations often fail to account for the varying strengths of different preferences or they rely on noisy margin information derived from preference ratings. Furthermore, many existing methods that use adaptive margins assume access t","authors_text":"Greg Durrett, Prasann Singhal, Scott Niekum, Yaswanth Chittepu","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-26T19:03:24Z","title":"Adaptive Margin RLHF via Preference over Preferences"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.22851","kind":"arxiv","version":4},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:64c3fc5763f1a24ad4e083df19fbc5856074217ffcbfe5bc90d648c84a1359d3","target":"record","created_at":"2026-07-07T00:15:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"07acff6c39f6c9cfa8cef061901f7c7e7c036c48208c6b27b0cc7a08b134c120","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-26T19:03:24Z","title_canon_sha256":"04cd33ead47ebe9916bae403c5a8daad273bf65ce2e58f7f4db244cd94382956"},"schema_version":"1.0","source":{"id":"2509.22851","kind":"arxiv","version":4}},"canonical_sha256":"dc1abd4b09b7a440a09d5986db7b7c4b7d98cc5b71506f6de777c5fa6802ba15","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"dc1abd4b09b7a440a09d5986db7b7c4b7d98cc5b71506f6de777c5fa6802ba15","first_computed_at":"2026-07-07T00:15:49.939184Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-07T00:15:49.939184Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"yqXLL7tP+CdLVhG0UNkslqigDT7s7ey6MfLpNw5fWcLnS1rosaI7SyXWxkPMXDXDo9lrDYzZitGW8m8ppM45CA==","signature_status":"signed_v1","signed_at":"2026-07-07T00:15:49.939906Z","signed_message":"canonical_sha256_bytes"},"source_id":"2509.22851","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:64c3fc5763f1a24ad4e083df19fbc5856074217ffcbfe5bc90d648c84a1359d3","sha256:a98ae3e53243afb0adbd235c485ca216ada71be2b0adac99e4d337c289cd0f8e"],"state_sha256":"2e4e28450434acdfb52bfc40a2143c79df7855b82a1166ab16d620570862e1f0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"59Q7vd8c+uzRa45Nhhc8EgpoQ8pQHOZWlZ/GJcoRklDoHJsOkRD4r3XpG4hFn0NhsOlldpPJchJdX1yYvPh3Dg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T01:30:35.470742Z","bundle_sha256":"3abcce53ea5ead15259c12c1b24ad1f9717785448e19b3cf0c3f8ae2f6590663"}}