{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:M5WGF5EU6Z3PF22RQHH6QMUQPB","short_pith_number":"pith:M5WGF5EU","canonical_record":{"source":{"id":"2607.11432","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","cross_cats_sorted":[],"title_canon_sha256":"5258153306871f9f462317da4517ad992ee35550dfe035005849d3c2878eff5a","abstract_canon_sha256":"e15b338ef9444a246d3b86064a0da2614403c4d795f2e38854e935cd4a7c44da"},"schema_version":"1.0"},"canonical_sha256":"676c62f494f676f2eb5181cfe83290784878317df8b52a5bb35c0d40a084462c","source":{"kind":"arxiv","id":"2607.11432","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.11432","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"arxiv_version","alias_value":"2607.11432v1","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.11432","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_12","alias_value":"M5WGF5EU6Z3P","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_16","alias_value":"M5WGF5EU6Z3PF22R","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_8","alias_value":"M5WGF5EU","created_at":"2026-07-14T02:22:02Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:M5WGF5EU6Z3PF22RQHH6QMUQPB","target":"record","payload":{"canonical_record":{"source":{"id":"2607.11432","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","cross_cats_sorted":[],"title_canon_sha256":"5258153306871f9f462317da4517ad992ee35550dfe035005849d3c2878eff5a","abstract_canon_sha256":"e15b338ef9444a246d3b86064a0da2614403c4d795f2e38854e935cd4a7c44da"},"schema_version":"1.0"},"canonical_sha256":"676c62f494f676f2eb5181cfe83290784878317df8b52a5bb35c0d40a084462c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T02:22:02.419873Z","signature_b64":"IR02FeApI+KnOeC06aoHv3thN50arLPAaKQ8M8P1lZ4dy73LaMaWn4iGWwtYSghFhG/4Byw/FS1a/ZdXEdp/Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"676c62f494f676f2eb5181cfe83290784878317df8b52a5bb35c0d40a084462c","last_reissued_at":"2026-07-14T02:22:02.419004Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T02:22:02.419004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.11432","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-14T02:22:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"g3LM08WFOYOcpJIbHoX4X38qt7kInFaSz5bftOpdjSqYG4mDgiLwY2l1nNxyuNLIxaW6b+BkpcyBFgNFEEs9Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T09:23:50.917852Z"},"content_sha256":"fc6c41f609d6e4c9046522309d60ea8692372d094eb11468f6f4234da5578646","schema_version":"1.0","event_id":"sha256:fc6c41f609d6e4c9046522309d60ea8692372d094eb11468f6f4234da5578646"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:M5WGF5EU6Z3PF22RQHH6QMUQPB","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Generalizing Preference-based Reinforcement Learning: a Rationality Model for Incomparability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alberto Maria Metelli, Leonardo Bianconi, Marco Mussi, Simone Drago","submitted_at":"2026-07-13T11:37:27Z","abstract_excerpt":"In this work, we study the reinforcement learning (RL) problem from pairwise trajectory comparisons provided by a human expert. We generalize preference-based RL by formalizing a novel setting in which the expert can also label trajectory pairs as incomparable, i.e., when neither trajectory dominates the other. We introduce the learning problem and the desiderata that its solution should satisfy. Then, we propose a novel Bradley-Terry-inspired rationality model that effectively captures incomparabilities and infers a multi-dimensional reward function, and we study its properties. We provide a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.11432","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.11432/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-14T02:22:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8/ct4lsrO2JcMB/VqYv7o8dw7qVSZfiShocZsanKe5+nJIXJmLunNSMuDr58nIIwdBrMsCcYyvEjFQTm5AlPAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T09:23:50.918694Z"},"content_sha256":"c4ebecbe75dfdb7a452b7a6895f74b811ea618af136dd3f640d0aacda97746c5","schema_version":"1.0","event_id":"sha256:c4ebecbe75dfdb7a452b7a6895f74b811ea618af136dd3f640d0aacda97746c5"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/M5WGF5EU6Z3PF22RQHH6QMUQPB/bundle.json","state_url":"https://pith.science/pith/M5WGF5EU6Z3PF22RQHH6QMUQPB/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/M5WGF5EU6Z3PF22RQHH6QMUQPB/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T09:23:50Z","links":{"resolver":"https://pith.science/pith/M5WGF5EU6Z3PF22RQHH6QMUQPB","bundle":"https://pith.science/pith/M5WGF5EU6Z3PF22RQHH6QMUQPB/bundle.json","state":"https://pith.science/pith/M5WGF5EU6Z3PF22RQHH6QMUQPB/state.json","well_known_bundle":"https://pith.science/.well-known/pith/M5WGF5EU6Z3PF22RQHH6QMUQPB/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:M5WGF5EU6Z3PF22RQHH6QMUQPB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e15b338ef9444a246d3b86064a0da2614403c4d795f2e38854e935cd4a7c44da","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","title_canon_sha256":"5258153306871f9f462317da4517ad992ee35550dfe035005849d3c2878eff5a"},"schema_version":"1.0","source":{"id":"2607.11432","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.11432","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"arxiv_version","alias_value":"2607.11432v1","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.11432","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_12","alias_value":"M5WGF5EU6Z3P","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_16","alias_value":"M5WGF5EU6Z3PF22R","created_at":"2026-07-14T02:22:02Z"},{"alias_kind":"pith_short_8","alias_value":"M5WGF5EU","created_at":"2026-07-14T02:22:02Z"}],"graph_snapshots":[{"event_id":"sha256:c4ebecbe75dfdb7a452b7a6895f74b811ea618af136dd3f640d0aacda97746c5","target":"graph","created_at":"2026-07-14T02:22:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.11432/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this work, we study the reinforcement learning (RL) problem from pairwise trajectory comparisons provided by a human expert. We generalize preference-based RL by formalizing a novel setting in which the expert can also label trajectory pairs as incomparable, i.e., when neither trajectory dominates the other. We introduce the learning problem and the desiderata that its solution should satisfy. Then, we propose a novel Bradley-Terry-inspired rationality model that effectively captures incomparabilities and infers a multi-dimensional reward function, and we study its properties. We provide a ","authors_text":"Alberto Maria Metelli, Leonardo Bianconi, Marco Mussi, Simone Drago","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","title":"Generalizing Preference-based Reinforcement Learning: a Rationality Model for Incomparability"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.11432","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fc6c41f609d6e4c9046522309d60ea8692372d094eb11468f6f4234da5578646","target":"record","created_at":"2026-07-14T02:22:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e15b338ef9444a246d3b86064a0da2614403c4d795f2e38854e935cd4a7c44da","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-13T11:37:27Z","title_canon_sha256":"5258153306871f9f462317da4517ad992ee35550dfe035005849d3c2878eff5a"},"schema_version":"1.0","source":{"id":"2607.11432","kind":"arxiv","version":1}},"canonical_sha256":"676c62f494f676f2eb5181cfe83290784878317df8b52a5bb35c0d40a084462c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"676c62f494f676f2eb5181cfe83290784878317df8b52a5bb35c0d40a084462c","first_computed_at":"2026-07-14T02:22:02.419004Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-14T02:22:02.419004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"IR02FeApI+KnOeC06aoHv3thN50arLPAaKQ8M8P1lZ4dy73LaMaWn4iGWwtYSghFhG/4Byw/FS1a/ZdXEdp/Dg==","signature_status":"signed_v1","signed_at":"2026-07-14T02:22:02.419873Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.11432","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fc6c41f609d6e4c9046522309d60ea8692372d094eb11468f6f4234da5578646","sha256:c4ebecbe75dfdb7a452b7a6895f74b811ea618af136dd3f640d0aacda97746c5"],"state_sha256":"b08f11a0f47afab108a658b29c10e02197c138be1f8df96c2f598965264fdfb0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Em1L9rWdiFhLOlrkC9eTWd2v6mcstL0Knt7v7oeMEpcuuLjPbJsFrLJm+VEmYRDtAGHODrml4nAW0XkETcqcDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T09:23:50.924705Z","bundle_sha256":"aad643ea6603f3649947dbedf037df583ef9416c48811b8b62b9dd68b566e881"}}