{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:W6RHWMTH7DVFRKHKDNE4A4CDXP","short_pith_number":"pith:W6RHWMTH","canonical_record":{"source":{"id":"2401.07181","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-14T01:09:48Z","cross_cats_sorted":[],"title_canon_sha256":"67e9ac5518a697ff3367b469342453b9caeaacc049e20f0f445400e7d2af89e3","abstract_canon_sha256":"7455f72e5f6907eb78c05c38a92084adacb58b603cad4b797f6be11b626ea687"},"schema_version":"1.0"},"canonical_sha256":"b7a27b3267f8ea58a8ea1b49c07043bbf01a4992b3f14a45b1ce0faad9bb359d","source":{"kind":"arxiv","id":"2401.07181","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.07181","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"arxiv_version","alias_value":"2401.07181v1","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.07181","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"pith_short_12","alias_value":"W6RHWMTH7DVF","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"pith_short_16","alias_value":"W6RHWMTH7DVFRKHK","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"pith_short_8","alias_value":"W6RHWMTH","created_at":"2026-07-05T07:33:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:W6RHWMTH7DVFRKHKDNE4A4CDXP","target":"record","payload":{"canonical_record":{"source":{"id":"2401.07181","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-14T01:09:48Z","cross_cats_sorted":[],"title_canon_sha256":"67e9ac5518a697ff3367b469342453b9caeaacc049e20f0f445400e7d2af89e3","abstract_canon_sha256":"7455f72e5f6907eb78c05c38a92084adacb58b603cad4b797f6be11b626ea687"},"schema_version":"1.0"},"canonical_sha256":"b7a27b3267f8ea58a8ea1b49c07043bbf01a4992b3f14a45b1ce0faad9bb359d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:33:33.211835Z","signature_b64":"i3a/8okgV1iNF7LSrzi2FFAWTrY9Uc+mx2mbR601LFfr2bWkimy4/s78bzfiLtQM9/MmND+VCQlp7mn+w9YXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7a27b3267f8ea58a8ea1b49c07043bbf01a4992b3f14a45b1ce0faad9bb359d","last_reissued_at":"2026-07-05T07:33:33.211297Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:33:33.211297Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.07181","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:33:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"37RGU3TYv9tqSpZVwHXmzYBnYYoHfTre5q5k+MWq+aywcmh0gQuNXjXz1I7Svc7UcPpwE4rI/yDQz6ddLFBBCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T16:52:06.742479Z"},"content_sha256":"4310a1da3a0dc9de56cedad7458db8ff43fb4c42453aaa5d3ac7b0bc1f3c81e7","schema_version":"1.0","event_id":"sha256:4310a1da3a0dc9de56cedad7458db8ff43fb4c42453aaa5d3ac7b0bc1f3c81e7"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:W6RHWMTH7DVFRKHKDNE4A4CDXP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reinforcement Learning from LLM Feedback to Counteract Goal Misgeneralization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Houda Nait El Barj, Theophile Sautory","submitted_at":"2024-01-14T01:09:48Z","abstract_excerpt":"We introduce a method to address goal misgeneralization in reinforcement learning (RL), leveraging Large Language Model (LLM) feedback during training. Goal misgeneralization, a type of robustness failure in RL occurs when an agent retains its capabilities out-of-distribution yet pursues a proxy rather than the intended one. Our approach utilizes LLMs to analyze an RL agent's policies during training and identify potential failure scenarios. The RL agent is then deployed in these scenarios, and a reward model is learnt through the LLM preferences and feedback. This LLM-informed reward model is"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.07181","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.07181/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:33:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"08N4Ae0FVV6s31sxUc2zVmqS9m0Gfztdq+BnI6EN9T0h3Bs/03JTqPee7euJ+xXB48WmtmGQBIFqCYs5bYWkCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T16:52:06.742978Z"},"content_sha256":"b7b43c6be4a46293dede903e47ff5560236fd62307f9bbeb759df1cf5139365a","schema_version":"1.0","event_id":"sha256:b7b43c6be4a46293dede903e47ff5560236fd62307f9bbeb759df1cf5139365a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/W6RHWMTH7DVFRKHKDNE4A4CDXP/bundle.json","state_url":"https://pith.science/pith/W6RHWMTH7DVFRKHKDNE4A4CDXP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/W6RHWMTH7DVFRKHKDNE4A4CDXP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T16:52:06Z","links":{"resolver":"https://pith.science/pith/W6RHWMTH7DVFRKHKDNE4A4CDXP","bundle":"https://pith.science/pith/W6RHWMTH7DVFRKHKDNE4A4CDXP/bundle.json","state":"https://pith.science/pith/W6RHWMTH7DVFRKHKDNE4A4CDXP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/W6RHWMTH7DVFRKHKDNE4A4CDXP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:W6RHWMTH7DVFRKHKDNE4A4CDXP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"7455f72e5f6907eb78c05c38a92084adacb58b603cad4b797f6be11b626ea687","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-14T01:09:48Z","title_canon_sha256":"67e9ac5518a697ff3367b469342453b9caeaacc049e20f0f445400e7d2af89e3"},"schema_version":"1.0","source":{"id":"2401.07181","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.07181","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"arxiv_version","alias_value":"2401.07181v1","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.07181","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"pith_short_12","alias_value":"W6RHWMTH7DVF","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"pith_short_16","alias_value":"W6RHWMTH7DVFRKHK","created_at":"2026-07-05T07:33:33Z"},{"alias_kind":"pith_short_8","alias_value":"W6RHWMTH","created_at":"2026-07-05T07:33:33Z"}],"graph_snapshots":[{"event_id":"sha256:b7b43c6be4a46293dede903e47ff5560236fd62307f9bbeb759df1cf5139365a","target":"graph","created_at":"2026-07-05T07:33:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.07181/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We introduce a method to address goal misgeneralization in reinforcement learning (RL), leveraging Large Language Model (LLM) feedback during training. Goal misgeneralization, a type of robustness failure in RL occurs when an agent retains its capabilities out-of-distribution yet pursues a proxy rather than the intended one. Our approach utilizes LLMs to analyze an RL agent's policies during training and identify potential failure scenarios. The RL agent is then deployed in these scenarios, and a reward model is learnt through the LLM preferences and feedback. This LLM-informed reward model is","authors_text":"Houda Nait El Barj, Theophile Sautory","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-14T01:09:48Z","title":"Reinforcement Learning from LLM Feedback to Counteract Goal Misgeneralization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.07181","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4310a1da3a0dc9de56cedad7458db8ff43fb4c42453aaa5d3ac7b0bc1f3c81e7","target":"record","created_at":"2026-07-05T07:33:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"7455f72e5f6907eb78c05c38a92084adacb58b603cad4b797f6be11b626ea687","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-14T01:09:48Z","title_canon_sha256":"67e9ac5518a697ff3367b469342453b9caeaacc049e20f0f445400e7d2af89e3"},"schema_version":"1.0","source":{"id":"2401.07181","kind":"arxiv","version":1}},"canonical_sha256":"b7a27b3267f8ea58a8ea1b49c07043bbf01a4992b3f14a45b1ce0faad9bb359d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b7a27b3267f8ea58a8ea1b49c07043bbf01a4992b3f14a45b1ce0faad9bb359d","first_computed_at":"2026-07-05T07:33:33.211297Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:33:33.211297Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"i3a/8okgV1iNF7LSrzi2FFAWTrY9Uc+mx2mbR601LFfr2bWkimy4/s78bzfiLtQM9/MmND+VCQlp7mn+w9YXAA==","signature_status":"signed_v1","signed_at":"2026-07-05T07:33:33.211835Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.07181","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4310a1da3a0dc9de56cedad7458db8ff43fb4c42453aaa5d3ac7b0bc1f3c81e7","sha256:b7b43c6be4a46293dede903e47ff5560236fd62307f9bbeb759df1cf5139365a"],"state_sha256":"3ccda944889c7e73c5c490dd61dce9bd7b79eeedbc8e1bcc0e778d3f8621679e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9gMAUm3an7VnEFIiXeIibWPu3XCASASRfd8AMRsUCoXCT1hKVP1oHkbIc67Kx6INvEYFihAjpLYfUqic9181Cg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T16:52:06.747355Z","bundle_sha256":"575b3bc8a64afc67ba83de557428a783bc48567d9739916aa8a92a9cab7f026d"}}