{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:YXA3Q5VNI53CON5UYP3EK45G6W","short_pith_number":"pith:YXA3Q5VN","canonical_record":{"source":{"id":"2402.11650","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-18T17:02:39Z","cross_cats_sorted":["cs.LO","cs.PL"],"title_canon_sha256":"b7e978e90096bd95113fd375e9ea80dda48735dff4fd824ee05c575f516e8b62","abstract_canon_sha256":"d0156b45ad9fdaf2be653c067b3c6b923cb30ec36e05ad0815285a7fc764adc8"},"schema_version":"1.0"},"canonical_sha256":"c5c1b876ad47762737b4c3f64573a6f596b85d2aaa9b877df8f60c224e386d40","source":{"kind":"arxiv","id":"2402.11650","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.11650","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"arxiv_version","alias_value":"2402.11650v2","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11650","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"pith_short_12","alias_value":"YXA3Q5VNI53C","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"pith_short_16","alias_value":"YXA3Q5VNI53CON5U","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"pith_short_8","alias_value":"YXA3Q5VN","created_at":"2026-07-05T09:59:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:YXA3Q5VNI53CON5UYP3EK45G6W","target":"record","payload":{"canonical_record":{"source":{"id":"2402.11650","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-18T17:02:39Z","cross_cats_sorted":["cs.LO","cs.PL"],"title_canon_sha256":"b7e978e90096bd95113fd375e9ea80dda48735dff4fd824ee05c575f516e8b62","abstract_canon_sha256":"d0156b45ad9fdaf2be653c067b3c6b923cb30ec36e05ad0815285a7fc764adc8"},"schema_version":"1.0"},"canonical_sha256":"c5c1b876ad47762737b4c3f64573a6f596b85d2aaa9b877df8f60c224e386d40","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:14.273279Z","signature_b64":"sWsv2T6apeEXeV/BbUVO80ftfnLohQXZ4/qM8ECXFd3BI8kSLCbiQSdWNL5eKjPcwmPTRO82Vi25VoS4FVG7Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5c1b876ad47762737b4c3f64573a6f596b85d2aaa9b877df8f60c224e386d40","last_reissued_at":"2026-07-05T09:59:14.272831Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:14.272831Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2402.11650","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:59:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9+XLqpvewV4UZ9Tur+t4oIh2HY3ishGFhxyKU/gbEwae6PX1iXdBriEG0EuVQsRAiFUq0k8KEhkgT+zeTsc+AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T23:08:48.476412Z"},"content_sha256":"808e0af1206ac7d03ca92128918bd7d7eeb02a27a07b065ec0666eb5797c2496","schema_version":"1.0","event_id":"sha256:808e0af1206ac7d03ca92128918bd7d7eeb02a27a07b065ec0666eb5797c2496"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:YXA3Q5VNI53CON5UYP3EK45G6W","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Programmatic Reinforcement Learning: Navigating Gridworlds","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LO","cs.PL"],"primary_cat":"cs.LG","authors_text":"Guruprerana Shabadi, Nathana\\\"el Fijalkow, Th\\'eo Matricon","submitted_at":"2024-02-18T17:02:39Z","abstract_excerpt":"The field of reinforcement learning (RL) is concerned with algorithms for learning optimal policies in unknown stochastic environments. Programmatic RL studies representations of policies as programs, meaning involving higher order constructs such as control loops. Despite attracting a lot of attention at the intersection of the machine learning and formal methods communities, very little is known on the theoretical front about programmatic RL: what are good classes of programmatic policies? How large are optimal programmatic policies? How can we learn them? The goal of this paper is to give f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11650","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:59:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PTeNheiR9KUdLat6zVvKjhQgMSc8/dcZYEJ60krQtiNzUEsDSOXYJTJEVhls8lnk/2z6NsDMPvzU00+FUSgXAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T23:08:48.480846Z"},"content_sha256":"8a8af6045bb21b409a318aeb9f74bd3323baab260e9fda7537c95c721aaf0813","schema_version":"1.0","event_id":"sha256:8a8af6045bb21b409a318aeb9f74bd3323baab260e9fda7537c95c721aaf0813"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YXA3Q5VNI53CON5UYP3EK45G6W/bundle.json","state_url":"https://pith.science/pith/YXA3Q5VNI53CON5UYP3EK45G6W/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YXA3Q5VNI53CON5UYP3EK45G6W/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T23:08:48Z","links":{"resolver":"https://pith.science/pith/YXA3Q5VNI53CON5UYP3EK45G6W","bundle":"https://pith.science/pith/YXA3Q5VNI53CON5UYP3EK45G6W/bundle.json","state":"https://pith.science/pith/YXA3Q5VNI53CON5UYP3EK45G6W/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YXA3Q5VNI53CON5UYP3EK45G6W/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:YXA3Q5VNI53CON5UYP3EK45G6W","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d0156b45ad9fdaf2be653c067b3c6b923cb30ec36e05ad0815285a7fc764adc8","cross_cats_sorted":["cs.LO","cs.PL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-18T17:02:39Z","title_canon_sha256":"b7e978e90096bd95113fd375e9ea80dda48735dff4fd824ee05c575f516e8b62"},"schema_version":"1.0","source":{"id":"2402.11650","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.11650","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"arxiv_version","alias_value":"2402.11650v2","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11650","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"pith_short_12","alias_value":"YXA3Q5VNI53C","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"pith_short_16","alias_value":"YXA3Q5VNI53CON5U","created_at":"2026-07-05T09:59:14Z"},{"alias_kind":"pith_short_8","alias_value":"YXA3Q5VN","created_at":"2026-07-05T09:59:14Z"}],"graph_snapshots":[{"event_id":"sha256:8a8af6045bb21b409a318aeb9f74bd3323baab260e9fda7537c95c721aaf0813","target":"graph","created_at":"2026-07-05T09:59:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2402.11650/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The field of reinforcement learning (RL) is concerned with algorithms for learning optimal policies in unknown stochastic environments. Programmatic RL studies representations of policies as programs, meaning involving higher order constructs such as control loops. Despite attracting a lot of attention at the intersection of the machine learning and formal methods communities, very little is known on the theoretical front about programmatic RL: what are good classes of programmatic policies? How large are optimal programmatic policies? How can we learn them? The goal of this paper is to give f","authors_text":"Guruprerana Shabadi, Nathana\\\"el Fijalkow, Th\\'eo Matricon","cross_cats":["cs.LO","cs.PL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-18T17:02:39Z","title":"Programmatic Reinforcement Learning: Navigating Gridworlds"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11650","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:808e0af1206ac7d03ca92128918bd7d7eeb02a27a07b065ec0666eb5797c2496","target":"record","created_at":"2026-07-05T09:59:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d0156b45ad9fdaf2be653c067b3c6b923cb30ec36e05ad0815285a7fc764adc8","cross_cats_sorted":["cs.LO","cs.PL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-18T17:02:39Z","title_canon_sha256":"b7e978e90096bd95113fd375e9ea80dda48735dff4fd824ee05c575f516e8b62"},"schema_version":"1.0","source":{"id":"2402.11650","kind":"arxiv","version":2}},"canonical_sha256":"c5c1b876ad47762737b4c3f64573a6f596b85d2aaa9b877df8f60c224e386d40","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c5c1b876ad47762737b4c3f64573a6f596b85d2aaa9b877df8f60c224e386d40","first_computed_at":"2026-07-05T09:59:14.272831Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:59:14.272831Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"sWsv2T6apeEXeV/BbUVO80ftfnLohQXZ4/qM8ECXFd3BI8kSLCbiQSdWNL5eKjPcwmPTRO82Vi25VoS4FVG7Dw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:59:14.273279Z","signed_message":"canonical_sha256_bytes"},"source_id":"2402.11650","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:808e0af1206ac7d03ca92128918bd7d7eeb02a27a07b065ec0666eb5797c2496","sha256:8a8af6045bb21b409a318aeb9f74bd3323baab260e9fda7537c95c721aaf0813"],"state_sha256":"1a71e9f16c1a6e89376a22c218ac1f1cd8c43c8215fc25bfa040c921b3c9d78b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fq2Mq5Zs67xcXRfBcN7euEarBVhpFwqhtWrprskObqXyVJ74yyOvacSyK9Vym04UgoTVLELD5//ACZLHBgfRCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T23:08:48.491199Z","bundle_sha256":"5a8686f8c6916d33ad1468b11579da0f141bf33bd3df4c711954507e420ebfdf"}}