{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:NLHP4ERB2Q42O5WJOQ4XNRZV5F","short_pith_number":"pith:NLHP4ERB","canonical_record":{"source":{"id":"2309.03651","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-07T11:46:57Z","cross_cats_sorted":[],"title_canon_sha256":"551aba4f4d6bedba60a32f72468d1b9cdfeb5180d0936826f1242313d844a757","abstract_canon_sha256":"5ed0687c4cb113c2bfbc4cbb70115d788951f5df28336519cefae351b4bee9bc"},"schema_version":"1.0"},"canonical_sha256":"6acefe1221d439a776c9743976c735e978cca058514b8f6bf9c004818b98b8ba","source":{"kind":"arxiv","id":"2309.03651","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2309.03651","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"arxiv_version","alias_value":"2309.03651v1","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.03651","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"pith_short_12","alias_value":"NLHP4ERB2Q42","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"pith_short_16","alias_value":"NLHP4ERB2Q42O5WJ","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"pith_short_8","alias_value":"NLHP4ERB","created_at":"2026-07-05T06:48:39Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:NLHP4ERB2Q42O5WJOQ4XNRZV5F","target":"record","payload":{"canonical_record":{"source":{"id":"2309.03651","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-07T11:46:57Z","cross_cats_sorted":[],"title_canon_sha256":"551aba4f4d6bedba60a32f72468d1b9cdfeb5180d0936826f1242313d844a757","abstract_canon_sha256":"5ed0687c4cb113c2bfbc4cbb70115d788951f5df28336519cefae351b4bee9bc"},"schema_version":"1.0"},"canonical_sha256":"6acefe1221d439a776c9743976c735e978cca058514b8f6bf9c004818b98b8ba","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:39.990103Z","signature_b64":"OfjvjaGYZNTxlgRm2r09Adp0a6CHvXQoXXJIkxjjdf+At26zpwtxMpN6Y+hd1p6XkumnJXE71RjmOAX4mLtVCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6acefe1221d439a776c9743976c735e978cca058514b8f6bf9c004818b98b8ba","last_reissued_at":"2026-07-05T06:48:39.989558Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:39.989558Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2309.03651","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:48:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"00oB774K2+FoN1JirUyZBW9KSBOBBI6E6xXmo2oZRI36oIDqXdYb+kmEcYuUWaghQj+d37RI+NVmv71OG9sgDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T08:27:36.951049Z"},"content_sha256":"48adcee7ba98bc7e1ae90a159804a7d11a563eb82c3796c28a12d886ec1cab93","schema_version":"1.0","event_id":"sha256:48adcee7ba98bc7e1ae90a159804a7d11a563eb82c3796c28a12d886ec1cab93"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:NLHP4ERB2Q42O5WJOQ4XNRZV5F","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Learning of Generalizable and Interpretable Knowledge in Grid-Based Reinforcement Learning Environments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Johannes Maucher, Manuel Eberhardinger, Setareh Maghsudi","submitted_at":"2023-09-07T11:46:57Z","abstract_excerpt":"Understanding the interactions of agents trained with deep reinforcement learning is crucial for deploying agents in games or the real world. In the former, unreasonable actions confuse players. In the latter, that effect is even more significant, as unexpected behavior cause accidents with potentially grave and long-lasting consequences for the involved individuals. In this work, we propose using program synthesis to imitate reinforcement learning policies after seeing a trajectory of the action sequence. Programs have the advantage that they are inherently interpretable and verifiable for co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.03651","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.03651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:48:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fU79pZ8s5mgZHB7vB7feRP2m6TyNj7smGR8QDy/fPVVjF6h53d6jfprc886o5llPWalIzVCF2QqdUlxteW5/CQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T08:27:36.951800Z"},"content_sha256":"e1400e21b5a40bb68aeb8804fd60c43b11b1e45397be025d48d534c3f4e1cce8","schema_version":"1.0","event_id":"sha256:e1400e21b5a40bb68aeb8804fd60c43b11b1e45397be025d48d534c3f4e1cce8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NLHP4ERB2Q42O5WJOQ4XNRZV5F/bundle.json","state_url":"https://pith.science/pith/NLHP4ERB2Q42O5WJOQ4XNRZV5F/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NLHP4ERB2Q42O5WJOQ4XNRZV5F/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T08:27:36Z","links":{"resolver":"https://pith.science/pith/NLHP4ERB2Q42O5WJOQ4XNRZV5F","bundle":"https://pith.science/pith/NLHP4ERB2Q42O5WJOQ4XNRZV5F/bundle.json","state":"https://pith.science/pith/NLHP4ERB2Q42O5WJOQ4XNRZV5F/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NLHP4ERB2Q42O5WJOQ4XNRZV5F/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:NLHP4ERB2Q42O5WJOQ4XNRZV5F","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5ed0687c4cb113c2bfbc4cbb70115d788951f5df28336519cefae351b4bee9bc","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-07T11:46:57Z","title_canon_sha256":"551aba4f4d6bedba60a32f72468d1b9cdfeb5180d0936826f1242313d844a757"},"schema_version":"1.0","source":{"id":"2309.03651","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2309.03651","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"arxiv_version","alias_value":"2309.03651v1","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.03651","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"pith_short_12","alias_value":"NLHP4ERB2Q42","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"pith_short_16","alias_value":"NLHP4ERB2Q42O5WJ","created_at":"2026-07-05T06:48:39Z"},{"alias_kind":"pith_short_8","alias_value":"NLHP4ERB","created_at":"2026-07-05T06:48:39Z"}],"graph_snapshots":[{"event_id":"sha256:e1400e21b5a40bb68aeb8804fd60c43b11b1e45397be025d48d534c3f4e1cce8","target":"graph","created_at":"2026-07-05T06:48:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2309.03651/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Understanding the interactions of agents trained with deep reinforcement learning is crucial for deploying agents in games or the real world. In the former, unreasonable actions confuse players. In the latter, that effect is even more significant, as unexpected behavior cause accidents with potentially grave and long-lasting consequences for the involved individuals. In this work, we propose using program synthesis to imitate reinforcement learning policies after seeing a trajectory of the action sequence. Programs have the advantage that they are inherently interpretable and verifiable for co","authors_text":"Johannes Maucher, Manuel Eberhardinger, Setareh Maghsudi","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-07T11:46:57Z","title":"Learning of Generalizable and Interpretable Knowledge in Grid-Based Reinforcement Learning Environments"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.03651","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:48adcee7ba98bc7e1ae90a159804a7d11a563eb82c3796c28a12d886ec1cab93","target":"record","created_at":"2026-07-05T06:48:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5ed0687c4cb113c2bfbc4cbb70115d788951f5df28336519cefae351b4bee9bc","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-07T11:46:57Z","title_canon_sha256":"551aba4f4d6bedba60a32f72468d1b9cdfeb5180d0936826f1242313d844a757"},"schema_version":"1.0","source":{"id":"2309.03651","kind":"arxiv","version":1}},"canonical_sha256":"6acefe1221d439a776c9743976c735e978cca058514b8f6bf9c004818b98b8ba","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6acefe1221d439a776c9743976c735e978cca058514b8f6bf9c004818b98b8ba","first_computed_at":"2026-07-05T06:48:39.989558Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:48:39.989558Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"OfjvjaGYZNTxlgRm2r09Adp0a6CHvXQoXXJIkxjjdf+At26zpwtxMpN6Y+hd1p6XkumnJXE71RjmOAX4mLtVCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T06:48:39.990103Z","signed_message":"canonical_sha256_bytes"},"source_id":"2309.03651","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:48adcee7ba98bc7e1ae90a159804a7d11a563eb82c3796c28a12d886ec1cab93","sha256:e1400e21b5a40bb68aeb8804fd60c43b11b1e45397be025d48d534c3f4e1cce8"],"state_sha256":"04b247a872948a9e5f5bf0b42d299a5d7a361b999b92e15d7ac4103e5c098df0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"d9GnCXf3QMaCW+xqZqCFlKBlveGTOaXmlREzjuLLnOF/FbqImDSvxzhreI1cOW7nsZTe2k67qgj8SUzY0frPBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T08:27:36.957026Z","bundle_sha256":"89f94413575fb058684c86c9d848c10b8dc5d6703a2df7bed0e00a9798ace438"}}