{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:5GJC53IINUAQCA5L5NDX57GDC7","short_pith_number":"pith:5GJC53II","canonical_record":{"source":{"id":"2502.02921","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T06:30:14Z","cross_cats_sorted":[],"title_canon_sha256":"1e650bd0539c1a9bf02b595f1251d67c1a8df06298424460c7ea77797b201dd9","abstract_canon_sha256":"59b8ab108c855e3008edf682086d790e686bb471abe4a74c07084c1a853d4250"},"schema_version":"1.0"},"canonical_sha256":"e9922eed086d010103abeb477efcc317ed73a9e68c5073749187841270a0503f","source":{"kind":"arxiv","id":"2502.02921","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.02921","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"arxiv_version","alias_value":"2502.02921v3","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02921","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"pith_short_12","alias_value":"5GJC53IINUAQ","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"pith_short_16","alias_value":"5GJC53IINUAQCA5L","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"pith_short_8","alias_value":"5GJC53II","created_at":"2026-07-05T11:10:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:5GJC53IINUAQCA5L5NDX57GDC7","target":"record","payload":{"canonical_record":{"source":{"id":"2502.02921","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T06:30:14Z","cross_cats_sorted":[],"title_canon_sha256":"1e650bd0539c1a9bf02b595f1251d67c1a8df06298424460c7ea77797b201dd9","abstract_canon_sha256":"59b8ab108c855e3008edf682086d790e686bb471abe4a74c07084c1a853d4250"},"schema_version":"1.0"},"canonical_sha256":"e9922eed086d010103abeb477efcc317ed73a9e68c5073749187841270a0503f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:51.308947Z","signature_b64":"8cFGCq/mU4uNpYNOI/bFGlnmzbE4SCIhCwTzUmUGLjIHgzQd7MUBPDM0h/6dtgOfDXuGrsN3TiYuXNBV2NqRCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e9922eed086d010103abeb477efcc317ed73a9e68c5073749187841270a0503f","last_reissued_at":"2026-07-05T11:10:51.308445Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:51.308445Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.02921","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:10:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"U6jhtSsgDXl2eVGKratz6KqFAT/ZfyJLJ2zhAgTkNVJIgk15odfVCSYBuNfrJEorcGuTqeDBv74XGnW2g5VxCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T04:13:12.716935Z"},"content_sha256":"c884091a55ba7c1e78ff2de3ca5b6e8e35982101f108bc33893d5d0389473542","schema_version":"1.0","event_id":"sha256:c884091a55ba7c1e78ff2de3ca5b6e8e35982101f108bc33893d5d0389473542"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:5GJC53IINUAQCA5L5NDX57GDC7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Robust Reward Alignment via Hypothesis Space Batch Cutting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Haode Zhang, Wanxin Jin, Yizhe Feng, Zhixian Xie","submitted_at":"2025-02-05T06:30:14Z","abstract_excerpt":"Reward design in reinforcement learning and optimal control is challenging. Preference-based alignment addresses this by enabling agents to learn rewards from ranked trajectory pairs provided by humans. However, existing methods often struggle from poor robustness to unknown false human preferences. In this work, we propose a robust and efficient reward alignment method based on a novel and geometrically interpretable perspective: hypothesis space batched cutting. Our method iteratively refines the reward hypothesis space through \"cuts\" based on batches of human preferences. Within each batch,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02921","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02921/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:10:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"caX/sIpU7ar6dJ9jAhQOptIGDe2ETTKFEPgJPMEcPeMwNrJ289Upz3fOO9hLlynZo1TdYhMfTS0iJb+VP63mAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T04:13:12.717445Z"},"content_sha256":"0efeb606c66d90b31c64245beb50a91a71789c5496f96e393cc496650073eef0","schema_version":"1.0","event_id":"sha256:0efeb606c66d90b31c64245beb50a91a71789c5496f96e393cc496650073eef0"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/5GJC53IINUAQCA5L5NDX57GDC7/bundle.json","state_url":"https://pith.science/pith/5GJC53IINUAQCA5L5NDX57GDC7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/5GJC53IINUAQCA5L5NDX57GDC7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T04:13:12Z","links":{"resolver":"https://pith.science/pith/5GJC53IINUAQCA5L5NDX57GDC7","bundle":"https://pith.science/pith/5GJC53IINUAQCA5L5NDX57GDC7/bundle.json","state":"https://pith.science/pith/5GJC53IINUAQCA5L5NDX57GDC7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/5GJC53IINUAQCA5L5NDX57GDC7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:5GJC53IINUAQCA5L5NDX57GDC7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"59b8ab108c855e3008edf682086d790e686bb471abe4a74c07084c1a853d4250","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T06:30:14Z","title_canon_sha256":"1e650bd0539c1a9bf02b595f1251d67c1a8df06298424460c7ea77797b201dd9"},"schema_version":"1.0","source":{"id":"2502.02921","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.02921","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"arxiv_version","alias_value":"2502.02921v3","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02921","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"pith_short_12","alias_value":"5GJC53IINUAQ","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"pith_short_16","alias_value":"5GJC53IINUAQCA5L","created_at":"2026-07-05T11:10:51Z"},{"alias_kind":"pith_short_8","alias_value":"5GJC53II","created_at":"2026-07-05T11:10:51Z"}],"graph_snapshots":[{"event_id":"sha256:0efeb606c66d90b31c64245beb50a91a71789c5496f96e393cc496650073eef0","target":"graph","created_at":"2026-07-05T11:10:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.02921/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward design in reinforcement learning and optimal control is challenging. Preference-based alignment addresses this by enabling agents to learn rewards from ranked trajectory pairs provided by humans. However, existing methods often struggle from poor robustness to unknown false human preferences. In this work, we propose a robust and efficient reward alignment method based on a novel and geometrically interpretable perspective: hypothesis space batched cutting. Our method iteratively refines the reward hypothesis space through \"cuts\" based on batches of human preferences. Within each batch,","authors_text":"Haode Zhang, Wanxin Jin, Yizhe Feng, Zhixian Xie","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T06:30:14Z","title":"Robust Reward Alignment via Hypothesis Space Batch Cutting"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02921","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c884091a55ba7c1e78ff2de3ca5b6e8e35982101f108bc33893d5d0389473542","target":"record","created_at":"2026-07-05T11:10:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"59b8ab108c855e3008edf682086d790e686bb471abe4a74c07084c1a853d4250","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T06:30:14Z","title_canon_sha256":"1e650bd0539c1a9bf02b595f1251d67c1a8df06298424460c7ea77797b201dd9"},"schema_version":"1.0","source":{"id":"2502.02921","kind":"arxiv","version":3}},"canonical_sha256":"e9922eed086d010103abeb477efcc317ed73a9e68c5073749187841270a0503f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e9922eed086d010103abeb477efcc317ed73a9e68c5073749187841270a0503f","first_computed_at":"2026-07-05T11:10:51.308445Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:10:51.308445Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"8cFGCq/mU4uNpYNOI/bFGlnmzbE4SCIhCwTzUmUGLjIHgzQd7MUBPDM0h/6dtgOfDXuGrsN3TiYuXNBV2NqRCA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:10:51.308947Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.02921","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c884091a55ba7c1e78ff2de3ca5b6e8e35982101f108bc33893d5d0389473542","sha256:0efeb606c66d90b31c64245beb50a91a71789c5496f96e393cc496650073eef0"],"state_sha256":"67e16a93a7db0cfa7b75b414e770daafcacc954459d2aa335eb3ea482a69805e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zt/FF082wxHO4f4HJHsZyhn6i+xc03XkusvPgFvGTqrWfEMT8wxglROo3nC6pVpr+5AU+Sha1Nvh9lDHjRhtDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T04:13:12.722983Z","bundle_sha256":"54730b94f3a4840c13a1081ecfaddc213a66be31ba4ee6d418e947c68ad4320e"}}