{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3GFDT742PWQXCSS6HJ6CDAJEPR","short_pith_number":"pith:3GFDT742","schema_version":"1.0","canonical_sha256":"d98a39ff9a7da1714a5e3a7c2181247c60c4ece1e38e8596f77db04339e707fe","source":{"kind":"arxiv","id":"2112.15236","version":1},"attestation_state":"computed","paper":{"title":"Learning Agent State Online with Recurrent Generate-and-Test","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amir Samani, Richard S. Sutton","submitted_at":"2021-12-30T23:10:32Z","abstract_excerpt":"Learning continually and online from a continuous stream of data is challenging, especially for a reinforcement learning agent with sequential data. When the environment only provides observations giving partial information about the state of the environment, the agent must learn the agent state based on the data stream of experience. We refer to the state learned directly from the data stream of experience as the agent state. Recurrent neural networks can learn the agent state, but the training methods are computationally expensive and sensitive to the hyper-parameters, making them unideal fo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.15236","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-30T23:10:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4b6b8fa142abbd819dd4f438b86c77e52de4192a82d21d7cb88e50ddede3f709","abstract_canon_sha256":"6e53ef8d1b114b5f592de27c5b59c02bcafcc296bbb47cbfea0e682af670677b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:44:37.814255Z","signature_b64":"MyJr6Pv8URc8Q6cgkzLgehzSeJLOxty5jfLaX650AEdnN4Lc3JQgSzsuFHTt+ggUdCKR9KOwFdECxPTBRTYcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d98a39ff9a7da1714a5e3a7c2181247c60c4ece1e38e8596f77db04339e707fe","last_reissued_at":"2026-07-05T03:44:37.813835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:44:37.813835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Agent State Online with Recurrent Generate-and-Test","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amir Samani, Richard S. Sutton","submitted_at":"2021-12-30T23:10:32Z","abstract_excerpt":"Learning continually and online from a continuous stream of data is challenging, especially for a reinforcement learning agent with sequential data. When the environment only provides observations giving partial information about the state of the environment, the agent must learn the agent state based on the data stream of experience. We refer to the state learned directly from the data stream of experience as the agent state. Recurrent neural networks can learn the agent state, but the training methods are computationally expensive and sensitive to the hyper-parameters, making them unideal fo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.15236","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.15236/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.15236","created_at":"2026-07-05T03:44:37.813888+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.15236v1","created_at":"2026-07-05T03:44:37.813888+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.15236","created_at":"2026-07-05T03:44:37.813888+00:00"},{"alias_kind":"pith_short_12","alias_value":"3GFDT742PWQX","created_at":"2026-07-05T03:44:37.813888+00:00"},{"alias_kind":"pith_short_16","alias_value":"3GFDT742PWQXCSS6","created_at":"2026-07-05T03:44:37.813888+00:00"},{"alias_kind":"pith_short_8","alias_value":"3GFDT742","created_at":"2026-07-05T03:44:37.813888+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16318","citing_title":"Investigating Action Encodings in Recurrent Neural Networks in Reinforcement Learning","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR","json":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR.json","graph_json":"https://pith.science/api/pith-number/3GFDT742PWQXCSS6HJ6CDAJEPR/graph.json","events_json":"https://pith.science/api/pith-number/3GFDT742PWQXCSS6HJ6CDAJEPR/events.json","paper":"https://pith.science/paper/3GFDT742"},"agent_actions":{"view_html":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR","download_json":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR.json","view_paper":"https://pith.science/paper/3GFDT742","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.15236&json=true","fetch_graph":"https://pith.science/api/pith-number/3GFDT742PWQXCSS6HJ6CDAJEPR/graph.json","fetch_events":"https://pith.science/api/pith-number/3GFDT742PWQXCSS6HJ6CDAJEPR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR/action/storage_attestation","attest_author":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR/action/author_attestation","sign_citation":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR/action/citation_signature","submit_replication":"https://pith.science/pith/3GFDT742PWQXCSS6HJ6CDAJEPR/action/replication_record"}},"created_at":"2026-07-05T03:44:37.813888+00:00","updated_at":"2026-07-05T03:44:37.813888+00:00"}