{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:MQIHFG5ZEA64S76M25YWV7ZW3Q","short_pith_number":"pith:MQIHFG5Z","canonical_record":{"source":{"id":"2501.07755","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-13T23:56:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2647de28ae083e096c019740cb8ccd3274108f7dc00a6a87cfe3676e1a0ee864","abstract_canon_sha256":"2e745cf5fcd243e21e9d8939490fafdc11c1f18c6f0b4a57a15f4fe435afa3d7"},"schema_version":"1.0"},"canonical_sha256":"6410729bb9203dc97fccd7716aff36dc0f9156f5736ace2e142e0a8669748c0f","source":{"kind":"arxiv","id":"2501.07755","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.07755","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"arxiv_version","alias_value":"2501.07755v1","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.07755","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"pith_short_12","alias_value":"MQIHFG5ZEA64","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"pith_short_16","alias_value":"MQIHFG5ZEA64S76M","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"pith_short_8","alias_value":"MQIHFG5Z","created_at":"2026-07-05T10:00:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:MQIHFG5ZEA64S76M25YWV7ZW3Q","target":"record","payload":{"canonical_record":{"source":{"id":"2501.07755","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-13T23:56:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2647de28ae083e096c019740cb8ccd3274108f7dc00a6a87cfe3676e1a0ee864","abstract_canon_sha256":"2e745cf5fcd243e21e9d8939490fafdc11c1f18c6f0b4a57a15f4fe435afa3d7"},"schema_version":"1.0"},"canonical_sha256":"6410729bb9203dc97fccd7716aff36dc0f9156f5736ace2e142e0a8669748c0f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:49.570063Z","signature_b64":"bws2Q64LKV+/NOQ4VqSDriBXSvU/K72OjQs2sGMwUhaH5xKbbIjGXy6qpyisU3wsv+QbwS6r8cRDh8CuF95NBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6410729bb9203dc97fccd7716aff36dc0f9156f5736ace2e142e0a8669748c0f","last_reissued_at":"2026-07-05T10:00:49.569661Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:49.569661Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2501.07755","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:00:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"d1mS5sQWX2dY0P6tVKnjJufj4FctqUTAU3SBgethkuVWHoQ1ADOkc6FQS+p9+pFvOIKKFQ/RNAGjU5U0EK/MDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T19:10:13.462413Z"},"content_sha256":"09b707b8356dcd5c55360a500591a361498b5b99bd2fbf060394fa7765ba4618","schema_version":"1.0","event_id":"sha256:09b707b8356dcd5c55360a500591a361498b5b99bd2fbf060394fa7765ba4618"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:MQIHFG5ZEA64S76M25YWV7ZW3Q","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Performance Optimization of Ratings-Based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Devin White, Evelyn Rose, Mingkang Wu, Nicholas R. Waytowich, Vernon Lawhern, Yongcan Cao","submitted_at":"2025-01-13T23:56:24Z","abstract_excerpt":"This paper explores multiple optimization methods to improve the performance of rating-based reinforcement learning (RbRL). RbRL, a method based on the idea of human ratings, has been developed to infer reward functions in reward-free environments for the subsequent policy learning via standard reinforcement learning, which requires the availability of reward functions. Specifically, RbRL minimizes the cross entropy loss that quantifies the differences between human ratings and estimated ratings derived from the inferred reward. Hence, a low loss means a high degree of consistency between huma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.07755","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.07755/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:00:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"3wyebZ0vfcOxb+sYBu3GliSzKSdnw7+U/cYAia8AWjAqs1wF+u9VdbY8Dn/PssxlYka8/nzcNF4NQEwLYJTPAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T19:10:13.463152Z"},"content_sha256":"c40c2fb954929c88eea6b6ebf60dec5c8f86ee644a8a66a39f4ce83feda3eb1b","schema_version":"1.0","event_id":"sha256:c40c2fb954929c88eea6b6ebf60dec5c8f86ee644a8a66a39f4ce83feda3eb1b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/MQIHFG5ZEA64S76M25YWV7ZW3Q/bundle.json","state_url":"https://pith.science/pith/MQIHFG5ZEA64S76M25YWV7ZW3Q/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/MQIHFG5ZEA64S76M25YWV7ZW3Q/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-20T19:10:13Z","links":{"resolver":"https://pith.science/pith/MQIHFG5ZEA64S76M25YWV7ZW3Q","bundle":"https://pith.science/pith/MQIHFG5ZEA64S76M25YWV7ZW3Q/bundle.json","state":"https://pith.science/pith/MQIHFG5ZEA64S76M25YWV7ZW3Q/state.json","well_known_bundle":"https://pith.science/.well-known/pith/MQIHFG5ZEA64S76M25YWV7ZW3Q/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:MQIHFG5ZEA64S76M25YWV7ZW3Q","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2e745cf5fcd243e21e9d8939490fafdc11c1f18c6f0b4a57a15f4fe435afa3d7","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-13T23:56:24Z","title_canon_sha256":"2647de28ae083e096c019740cb8ccd3274108f7dc00a6a87cfe3676e1a0ee864"},"schema_version":"1.0","source":{"id":"2501.07755","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.07755","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"arxiv_version","alias_value":"2501.07755v1","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.07755","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"pith_short_12","alias_value":"MQIHFG5ZEA64","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"pith_short_16","alias_value":"MQIHFG5ZEA64S76M","created_at":"2026-07-05T10:00:49Z"},{"alias_kind":"pith_short_8","alias_value":"MQIHFG5Z","created_at":"2026-07-05T10:00:49Z"}],"graph_snapshots":[{"event_id":"sha256:c40c2fb954929c88eea6b6ebf60dec5c8f86ee644a8a66a39f4ce83feda3eb1b","target":"graph","created_at":"2026-07-05T10:00:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2501.07755/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper explores multiple optimization methods to improve the performance of rating-based reinforcement learning (RbRL). RbRL, a method based on the idea of human ratings, has been developed to infer reward functions in reward-free environments for the subsequent policy learning via standard reinforcement learning, which requires the availability of reward functions. Specifically, RbRL minimizes the cross entropy loss that quantifies the differences between human ratings and estimated ratings derived from the inferred reward. Hence, a low loss means a high degree of consistency between huma","authors_text":"Devin White, Evelyn Rose, Mingkang Wu, Nicholas R. Waytowich, Vernon Lawhern, Yongcan Cao","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-13T23:56:24Z","title":"Performance Optimization of Ratings-Based Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.07755","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:09b707b8356dcd5c55360a500591a361498b5b99bd2fbf060394fa7765ba4618","target":"record","created_at":"2026-07-05T10:00:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2e745cf5fcd243e21e9d8939490fafdc11c1f18c6f0b4a57a15f4fe435afa3d7","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-13T23:56:24Z","title_canon_sha256":"2647de28ae083e096c019740cb8ccd3274108f7dc00a6a87cfe3676e1a0ee864"},"schema_version":"1.0","source":{"id":"2501.07755","kind":"arxiv","version":1}},"canonical_sha256":"6410729bb9203dc97fccd7716aff36dc0f9156f5736ace2e142e0a8669748c0f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6410729bb9203dc97fccd7716aff36dc0f9156f5736ace2e142e0a8669748c0f","first_computed_at":"2026-07-05T10:00:49.569661Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:00:49.569661Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"bws2Q64LKV+/NOQ4VqSDriBXSvU/K72OjQs2sGMwUhaH5xKbbIjGXy6qpyisU3wsv+QbwS6r8cRDh8CuF95NBg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:00:49.570063Z","signed_message":"canonical_sha256_bytes"},"source_id":"2501.07755","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:09b707b8356dcd5c55360a500591a361498b5b99bd2fbf060394fa7765ba4618","sha256:c40c2fb954929c88eea6b6ebf60dec5c8f86ee644a8a66a39f4ce83feda3eb1b"],"state_sha256":"ce93df6a76458bc62780dd52ae2193ac9b539271c766b89b2234fb40a83e93ca"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gc08bZjf/YEqgu+4o+0XfnDABcgI3Xy1BDnUZ2k+iWRfkfeMwI1WO3fx2jDgQqOAwQ/kLGdRwoYaVDSg6pcUCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-20T19:10:13.469411Z","bundle_sha256":"b684a56af0a1db0e2f17114acc24970ea719fab3070a087ae6f6974f1b6fb7c6"}}