{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:PARTMM2PPWT7TQPQNHOLEGWN4L","short_pith_number":"pith:PARTMM2P","canonical_record":{"source":{"id":"2203.05079","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-09T22:55:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cffd6c9d5ebc34577a49b19db1432b0e824467bff348baaeca147ead0817717d","abstract_canon_sha256":"2c5ad56c06580e3182a86f79cabbb7238fd45514e2aba4201d6efb45357b911e"},"schema_version":"1.0"},"canonical_sha256":"782336334f7da7f9c1f069dcb21acde2d7dcbb84544e5fe0278b0aee892b488d","source":{"kind":"arxiv","id":"2203.05079","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2203.05079","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"arxiv_version","alias_value":"2203.05079v1","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.05079","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"pith_short_12","alias_value":"PARTMM2PPWT7","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"pith_short_16","alias_value":"PARTMM2PPWT7TQPQ","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"pith_short_8","alias_value":"PARTMM2P","created_at":"2026-07-05T04:03:29Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:PARTMM2PPWT7TQPQNHOLEGWN4L","target":"record","payload":{"canonical_record":{"source":{"id":"2203.05079","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-09T22:55:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cffd6c9d5ebc34577a49b19db1432b0e824467bff348baaeca147ead0817717d","abstract_canon_sha256":"2c5ad56c06580e3182a86f79cabbb7238fd45514e2aba4201d6efb45357b911e"},"schema_version":"1.0"},"canonical_sha256":"782336334f7da7f9c1f069dcb21acde2d7dcbb84544e5fe0278b0aee892b488d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:03:29.536290Z","signature_b64":"BHiraiNGHEmUS/A4/WOrxTGoY8rPHViSkdAmyGA6F1V9RBsmtATju9vQJZa0nOyFn3HUVje3GjX/jhMSH9EPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"782336334f7da7f9c1f069dcb21acde2d7dcbb84544e5fe0278b0aee892b488d","last_reissued_at":"2026-07-05T04:03:29.535836Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:03:29.535836Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2203.05079","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:03:29Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AM7zVtUnSeU1TT9a3+gio/UkxXqBQOCsPgEhMuZtOpKJ0HSbUoJGNLM4ifFoL6KpombXWuVarLY2HuRU0x5tCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T16:17:11.182607Z"},"content_sha256":"26322aa8cfe60b450406e25012020769f402640409aa5451889ec8f007bb16d1","schema_version":"1.0","event_id":"sha256:26322aa8cfe60b450406e25012020769f402640409aa5451889ec8f007bb16d1"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:PARTMM2PPWT7TQPQNHOLEGWN4L","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"SAGE: Generating Symbolic Goals for Myopic Models in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrew Chester, Fabio Zambetta, John Thangarajah, Michael Dann","submitted_at":"2022-03-09T22:55:53Z","abstract_excerpt":"Model-based reinforcement learning algorithms are typically more sample efficient than their model-free counterparts, especially in sparse reward problems. Unfortunately, many interesting domains are too complex to specify the complete models required by traditional model-based approaches. Learning a model takes a large number of environment samples, and may not capture critical information if the environment is hard to explore. If we could specify an incomplete model and allow the agent to learn how best to use it, we could take advantage of our partial understanding of many domains. Existing"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.05079","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.05079/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:03:29Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rgD4uAdOv01MU93IFqiWJsrWSCvGSqh9dvrggI8nt7+9vnK29o0VZ8zmA0cYRyH6YQprd+uuDcltbcaoTg+MAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T16:17:11.184170Z"},"content_sha256":"2f16ae8299dd0dd5e0b032987972caf871348a268575c3f877f094bbf96d6f95","schema_version":"1.0","event_id":"sha256:2f16ae8299dd0dd5e0b032987972caf871348a268575c3f877f094bbf96d6f95"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/PARTMM2PPWT7TQPQNHOLEGWN4L/bundle.json","state_url":"https://pith.science/pith/PARTMM2PPWT7TQPQNHOLEGWN4L/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/PARTMM2PPWT7TQPQNHOLEGWN4L/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T16:17:11Z","links":{"resolver":"https://pith.science/pith/PARTMM2PPWT7TQPQNHOLEGWN4L","bundle":"https://pith.science/pith/PARTMM2PPWT7TQPQNHOLEGWN4L/bundle.json","state":"https://pith.science/pith/PARTMM2PPWT7TQPQNHOLEGWN4L/state.json","well_known_bundle":"https://pith.science/.well-known/pith/PARTMM2PPWT7TQPQNHOLEGWN4L/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:PARTMM2PPWT7TQPQNHOLEGWN4L","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2c5ad56c06580e3182a86f79cabbb7238fd45514e2aba4201d6efb45357b911e","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-09T22:55:53Z","title_canon_sha256":"cffd6c9d5ebc34577a49b19db1432b0e824467bff348baaeca147ead0817717d"},"schema_version":"1.0","source":{"id":"2203.05079","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2203.05079","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"arxiv_version","alias_value":"2203.05079v1","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.05079","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"pith_short_12","alias_value":"PARTMM2PPWT7","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"pith_short_16","alias_value":"PARTMM2PPWT7TQPQ","created_at":"2026-07-05T04:03:29Z"},{"alias_kind":"pith_short_8","alias_value":"PARTMM2P","created_at":"2026-07-05T04:03:29Z"}],"graph_snapshots":[{"event_id":"sha256:2f16ae8299dd0dd5e0b032987972caf871348a268575c3f877f094bbf96d6f95","target":"graph","created_at":"2026-07-05T04:03:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2203.05079/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Model-based reinforcement learning algorithms are typically more sample efficient than their model-free counterparts, especially in sparse reward problems. Unfortunately, many interesting domains are too complex to specify the complete models required by traditional model-based approaches. Learning a model takes a large number of environment samples, and may not capture critical information if the environment is hard to explore. If we could specify an incomplete model and allow the agent to learn how best to use it, we could take advantage of our partial understanding of many domains. Existing","authors_text":"Andrew Chester, Fabio Zambetta, John Thangarajah, Michael Dann","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-09T22:55:53Z","title":"SAGE: Generating Symbolic Goals for Myopic Models in Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.05079","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:26322aa8cfe60b450406e25012020769f402640409aa5451889ec8f007bb16d1","target":"record","created_at":"2026-07-05T04:03:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2c5ad56c06580e3182a86f79cabbb7238fd45514e2aba4201d6efb45357b911e","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-09T22:55:53Z","title_canon_sha256":"cffd6c9d5ebc34577a49b19db1432b0e824467bff348baaeca147ead0817717d"},"schema_version":"1.0","source":{"id":"2203.05079","kind":"arxiv","version":1}},"canonical_sha256":"782336334f7da7f9c1f069dcb21acde2d7dcbb84544e5fe0278b0aee892b488d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"782336334f7da7f9c1f069dcb21acde2d7dcbb84544e5fe0278b0aee892b488d","first_computed_at":"2026-07-05T04:03:29.535836Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:03:29.535836Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"BHiraiNGHEmUS/A4/WOrxTGoY8rPHViSkdAmyGA6F1V9RBsmtATju9vQJZa0nOyFn3HUVje3GjX/jhMSH9EPAA==","signature_status":"signed_v1","signed_at":"2026-07-05T04:03:29.536290Z","signed_message":"canonical_sha256_bytes"},"source_id":"2203.05079","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:26322aa8cfe60b450406e25012020769f402640409aa5451889ec8f007bb16d1","sha256:2f16ae8299dd0dd5e0b032987972caf871348a268575c3f877f094bbf96d6f95"],"state_sha256":"aedc40fe1b64305bcbf2dbd59ad0666b6fd31290f1f5120a91d198b73d9aebb8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NYayAq6UCCtCc+5G5M2NJzCDkPL0LEsd7rP1kMbjmpMccqXeQ/AE2YsvjOM1xjs2CrUGGldIC+4/Xuy9vSmzDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T16:17:11.193687Z","bundle_sha256":"f64c54040bc50f163bd56a4a5620c5802cf1062a5a3af85fb4d62df7fbdb2e78"}}