{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UWEEBRSO4RGUZWC3TCK7WUNIQI","short_pith_number":"pith:UWEEBRSO","schema_version":"1.0","canonical_sha256":"a58840c64ee44d4cd85b9895fb51a882220f627275062ad33e29f6aafa6a65ae","source":{"kind":"arxiv","id":"2312.11364","version":2},"attestation_state":"computed","paper":{"title":"Counting Reward Automata: Sample Efficient Reinforcement Learning Through the Exploitation of Reward Function Structure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Benjamin Rosman, Geraud Nangue Tasse, Steven James, Tristan Bester","submitted_at":"2023-12-18T17:20:38Z","abstract_excerpt":"We present counting reward automata-a finite state machine variant capable of modelling any reward function expressible as a formal language. Unlike previous approaches, which are limited to the expression of tasks as regular languages, our framework allows for tasks described by unrestricted grammars. We prove that an agent equipped with such an abstract machine is able to solve a larger set of tasks than those utilising current approaches. We show that this increase in expressive power does not come at the cost of increased automaton complexity. A selection of learning algorithms are present"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.11364","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-18T17:20:38Z","cross_cats_sorted":[],"title_canon_sha256":"7482a95d62c1afe0cb924a27560017f2ccb152c418ea5e6d11845e2cef63130a","abstract_canon_sha256":"d6989e80e45e67b1f080585e65f73dd94ffcdf65776a1a03b0d5c2fc9b7f8d5e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:46:28.959968Z","signature_b64":"Ul50opoNB3IfnIIxjaXauK8o7xcqr4PWE2/JzdD6I2TUXRV0y8WcLsdSFwpjG5H4OcxgrAiQqPoiR7zvoGdaDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a58840c64ee44d4cd85b9895fb51a882220f627275062ad33e29f6aafa6a65ae","last_reissued_at":"2026-07-05T07:46:28.959568Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:46:28.959568Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Counting Reward Automata: Sample Efficient Reinforcement Learning Through the Exploitation of Reward Function Structure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Benjamin Rosman, Geraud Nangue Tasse, Steven James, Tristan Bester","submitted_at":"2023-12-18T17:20:38Z","abstract_excerpt":"We present counting reward automata-a finite state machine variant capable of modelling any reward function expressible as a formal language. Unlike previous approaches, which are limited to the expression of tasks as regular languages, our framework allows for tasks described by unrestricted grammars. We prove that an agent equipped with such an abstract machine is able to solve a larger set of tasks than those utilising current approaches. We show that this increase in expressive power does not come at the cost of increased automaton complexity. A selection of learning algorithms are present"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.11364","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.11364/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.11364","created_at":"2026-07-05T07:46:28.959625+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.11364v2","created_at":"2026-07-05T07:46:28.959625+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.11364","created_at":"2026-07-05T07:46:28.959625+00:00"},{"alias_kind":"pith_short_12","alias_value":"UWEEBRSO4RGU","created_at":"2026-07-05T07:46:28.959625+00:00"},{"alias_kind":"pith_short_16","alias_value":"UWEEBRSO4RGUZWC3","created_at":"2026-07-05T07:46:28.959625+00:00"},{"alias_kind":"pith_short_8","alias_value":"UWEEBRSO","created_at":"2026-07-05T07:46:28.959625+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.05111","citing_title":"Reward Structure Shapes the Interaction Between Episodic Exploration and Neural Memory in Reinforcement Learning","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI","json":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI.json","graph_json":"https://pith.science/api/pith-number/UWEEBRSO4RGUZWC3TCK7WUNIQI/graph.json","events_json":"https://pith.science/api/pith-number/UWEEBRSO4RGUZWC3TCK7WUNIQI/events.json","paper":"https://pith.science/paper/UWEEBRSO"},"agent_actions":{"view_html":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI","download_json":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI.json","view_paper":"https://pith.science/paper/UWEEBRSO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.11364&json=true","fetch_graph":"https://pith.science/api/pith-number/UWEEBRSO4RGUZWC3TCK7WUNIQI/graph.json","fetch_events":"https://pith.science/api/pith-number/UWEEBRSO4RGUZWC3TCK7WUNIQI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI/action/storage_attestation","attest_author":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI/action/author_attestation","sign_citation":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI/action/citation_signature","submit_replication":"https://pith.science/pith/UWEEBRSO4RGUZWC3TCK7WUNIQI/action/replication_record"}},"created_at":"2026-07-05T07:46:28.959625+00:00","updated_at":"2026-07-05T07:46:28.959625+00:00"}