{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:2BKDJRJONHUIPUXM6JKU7S5NLZ","short_pith_number":"pith:2BKDJRJO","canonical_record":{"source":{"id":"2206.01078","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-02T15:04:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fb9cb4535af514b7110dbe933edb3130bb92948566ea6ce0f8e89f655d00e134","abstract_canon_sha256":"c5c7e4f4778eddb6653f998d199c2aa668f82325ae0842092bbefc9e702652ff"},"schema_version":"1.0"},"canonical_sha256":"d05434c52e69e887d2ecf2554fcbad5e592a1cd5413b85814f2fbc667f75339d","source":{"kind":"arxiv","id":"2206.01078","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.01078","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"arxiv_version","alias_value":"2206.01078v2","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.01078","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"pith_short_12","alias_value":"2BKDJRJONHUI","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"pith_short_16","alias_value":"2BKDJRJONHUIPUXM","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"pith_short_8","alias_value":"2BKDJRJO","created_at":"2026-07-05T05:14:57Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:2BKDJRJONHUIPUXM6JKU7S5NLZ","target":"record","payload":{"canonical_record":{"source":{"id":"2206.01078","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-02T15:04:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fb9cb4535af514b7110dbe933edb3130bb92948566ea6ce0f8e89f655d00e134","abstract_canon_sha256":"c5c7e4f4778eddb6653f998d199c2aa668f82325ae0842092bbefc9e702652ff"},"schema_version":"1.0"},"canonical_sha256":"d05434c52e69e887d2ecf2554fcbad5e592a1cd5413b85814f2fbc667f75339d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:14:57.707223Z","signature_b64":"9dTCc0xJbIO4+6Zyw/JJ0VklqmDqVy2FzmDt+ZwdtGw10E2YLMUPWsbhMYHu4qyepGH+SlDDAMbTWpT+9qBbCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d05434c52e69e887d2ecf2554fcbad5e592a1cd5413b85814f2fbc667f75339d","last_reissued_at":"2026-07-05T05:14:57.706772Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:14:57.706772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2206.01078","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:14:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"S65yi1OEPGhrZYyNUUJ57F8OVEMQQS0PRKkJRo6buZsi9+5fpt53HnKk5SwKm72OJHU10eNhvX954HmRBmWxDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T23:38:03.729898Z"},"content_sha256":"4c21c1f05b1141f7f7935e774196cd409fd3a5c6e0d78dd73de680d6078b0dbe","schema_version":"1.0","event_id":"sha256:4c21c1f05b1141f7f7935e774196cd409fd3a5c6e0d78dd73de680d6078b0dbe"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:2BKDJRJONHUIPUXM6JKU7S5NLZ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Deep Transformer Q-Networks for Partially Observable Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christopher Amato, Kevin Esslinger, Robert Platt","submitted_at":"2022-06-02T15:04:18Z","abstract_excerpt":"Real-world reinforcement learning tasks often involve some form of partial observability where the observations only give a partial or noisy view of the true state of the world. Such tasks typically require some form of memory, where the agent has access to multiple past observations, in order to perform well. One popular way to incorporate memory is by using a recurrent neural network to access the agent's history. However, recurrent neural networks in reinforcement learning are often fragile and difficult to train, susceptible to catastrophic forgetting and sometimes fail completely as a res"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.01078","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.01078/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:14:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ErA8IiGSKp9PZv04q8jqWLD3I+BjPSoIlx0Q9Y+csQJ/UhOD7/azihQR6fSnr6MVheZBi0WFzUbjgl/YBOm7Bw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T23:38:03.730437Z"},"content_sha256":"cdf3a628c0d844e3a75ec2021e9aab8d2458a275a2529d56de2e9fcdc5534973","schema_version":"1.0","event_id":"sha256:cdf3a628c0d844e3a75ec2021e9aab8d2458a275a2529d56de2e9fcdc5534973"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2BKDJRJONHUIPUXM6JKU7S5NLZ/bundle.json","state_url":"https://pith.science/pith/2BKDJRJONHUIPUXM6JKU7S5NLZ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2BKDJRJONHUIPUXM6JKU7S5NLZ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T23:38:03Z","links":{"resolver":"https://pith.science/pith/2BKDJRJONHUIPUXM6JKU7S5NLZ","bundle":"https://pith.science/pith/2BKDJRJONHUIPUXM6JKU7S5NLZ/bundle.json","state":"https://pith.science/pith/2BKDJRJONHUIPUXM6JKU7S5NLZ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2BKDJRJONHUIPUXM6JKU7S5NLZ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:2BKDJRJONHUIPUXM6JKU7S5NLZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c5c7e4f4778eddb6653f998d199c2aa668f82325ae0842092bbefc9e702652ff","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-02T15:04:18Z","title_canon_sha256":"fb9cb4535af514b7110dbe933edb3130bb92948566ea6ce0f8e89f655d00e134"},"schema_version":"1.0","source":{"id":"2206.01078","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.01078","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"arxiv_version","alias_value":"2206.01078v2","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.01078","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"pith_short_12","alias_value":"2BKDJRJONHUI","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"pith_short_16","alias_value":"2BKDJRJONHUIPUXM","created_at":"2026-07-05T05:14:57Z"},{"alias_kind":"pith_short_8","alias_value":"2BKDJRJO","created_at":"2026-07-05T05:14:57Z"}],"graph_snapshots":[{"event_id":"sha256:cdf3a628c0d844e3a75ec2021e9aab8d2458a275a2529d56de2e9fcdc5534973","target":"graph","created_at":"2026-07-05T05:14:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2206.01078/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Real-world reinforcement learning tasks often involve some form of partial observability where the observations only give a partial or noisy view of the true state of the world. Such tasks typically require some form of memory, where the agent has access to multiple past observations, in order to perform well. One popular way to incorporate memory is by using a recurrent neural network to access the agent's history. However, recurrent neural networks in reinforcement learning are often fragile and difficult to train, susceptible to catastrophic forgetting and sometimes fail completely as a res","authors_text":"Christopher Amato, Kevin Esslinger, Robert Platt","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-02T15:04:18Z","title":"Deep Transformer Q-Networks for Partially Observable Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.01078","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4c21c1f05b1141f7f7935e774196cd409fd3a5c6e0d78dd73de680d6078b0dbe","target":"record","created_at":"2026-07-05T05:14:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c5c7e4f4778eddb6653f998d199c2aa668f82325ae0842092bbefc9e702652ff","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-02T15:04:18Z","title_canon_sha256":"fb9cb4535af514b7110dbe933edb3130bb92948566ea6ce0f8e89f655d00e134"},"schema_version":"1.0","source":{"id":"2206.01078","kind":"arxiv","version":2}},"canonical_sha256":"d05434c52e69e887d2ecf2554fcbad5e592a1cd5413b85814f2fbc667f75339d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d05434c52e69e887d2ecf2554fcbad5e592a1cd5413b85814f2fbc667f75339d","first_computed_at":"2026-07-05T05:14:57.706772Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:14:57.706772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"9dTCc0xJbIO4+6Zyw/JJ0VklqmDqVy2FzmDt+ZwdtGw10E2YLMUPWsbhMYHu4qyepGH+SlDDAMbTWpT+9qBbCg==","signature_status":"signed_v1","signed_at":"2026-07-05T05:14:57.707223Z","signed_message":"canonical_sha256_bytes"},"source_id":"2206.01078","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4c21c1f05b1141f7f7935e774196cd409fd3a5c6e0d78dd73de680d6078b0dbe","sha256:cdf3a628c0d844e3a75ec2021e9aab8d2458a275a2529d56de2e9fcdc5534973"],"state_sha256":"7d5f12279869a63cb7c36aa3998153522daf2ef13d0c8493b96880fc7e8ce071"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XyqIfcbOEyDmYvnfVkXtbEgxTOjaULf2NTaEgHK1t3aBS0gHMzVN/0qX6SjkSzLM9oWv04fMBc3myK8r+XC4Bg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T23:38:03.735988Z","bundle_sha256":"12e6e4d8d368356c97ce50c1815b82b32f068126d49f76111f453700f4994be7"}}