{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:M5WGF5EU6Z3PF22RQHH6QMUQPB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e15b338ef9444a246d3b86064a0da2614403c4d795f2e38854e935cd4a7c44da","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","title_canon_sha256":"5258153306871f9f462317da4517ad992ee35550dfe035005849d3c2878eff5a"},"schema_version":"1.0","source":{"id":"2607.11432","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.11432","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"arxiv_version","alias_value":"2607.11432v1","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.11432","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_12","alias_value":"M5WGF5EU6Z3P","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_16","alias_value":"M5WGF5EU6Z3PF22R","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_8","alias_value":"M5WGF5EU","created_at":"2026-07-14T02:22:02Z"}],"graph_snapshots":[{"event_id":"sha256:c4ebecbe75dfdb7a452b7a6895f74b811ea618af136dd3f640d0aacda97746c5","target":"graph","created_at":"2026-07-14T02:22:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.11432/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this work, we study the reinforcement learning (RL) problem from pairwise trajectory comparisons provided by a human expert. We generalize preference-based RL by formalizing a novel setting in which the expert can also label trajectory pairs as incomparable, i.e., when neither trajectory dominates the other. We introduce the learning problem and the desiderata that its solution should satisfy. Then, we propose a novel Bradley-Terry-inspired rationality model that effectively captures incomparabilities and infers a multi-dimensional reward function, and we study its properties. We provide a ","authors_text":"Alberto Maria Metelli, Leonardo Bianconi, Marco Mussi, Simone Drago","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","title":"Generalizing Preference-based Reinforcement Learning: a Rationality Model for Incomparability"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.11432","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fc6c41f609d6e4c9046522309d60ea8692372d094eb11468f6f4234da5578646","target":"record","created_at":"2026-07-14T02:22:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e15b338ef9444a246d3b86064a0da2614403c4d795f2e38854e935cd4a7c44da","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","title_canon_sha256":"5258153306871f9f462317da4517ad992ee35550dfe035005849d3c2878eff5a"},"schema_version":"1.0","source":{"id":"2607.11432","kind":"arxiv","version":1}},"canonical_sha256":"676c62f494f676f2eb5181cfe83290784878317df8b52a5bb35c0d40a084462c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"676c62f494f676f2eb5181cfe83290784878317df8b52a5bb35c0d40a084462c","first_computed_at":"2026-07-14T02:22:02.419004Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-14T02:22:02.419004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"IR02FeApI+KnOeC06aoHv3thN50arLPAaKQ8M8P1lZ4dy73LaMaWn4iGWwtYSghFhG/4Byw/FS1a/ZdXEdp/Dg==","signature_status":"signed_v1","signed_at":"2026-07-14T02:22:02.419873Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.11432","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fc6c41f609d6e4c9046522309d60ea8692372d094eb11468f6f4234da5578646","sha256:c4ebecbe75dfdb7a452b7a6895f74b811ea618af136dd3f640d0aacda97746c5"],"state_sha256":"b08f11a0f47afab108a658b29c10e02197c138be1f8df96c2f598965264fdfb0"}