{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JYNJBFPIALWAWE3Q6SEZ7GXDMP","short_pith_number":"pith:JYNJBFPI","schema_version":"1.0","canonical_sha256":"4e1a9095e802ec0b1370f4899f9ae363d189c4b59f05f49edafa83fbc1c7929c","source":{"kind":"arxiv","id":"2510.27329","version":2},"attestation_state":"computed","paper":{"title":"Reinforcement Learning for Long-Horizon Unordered Tasks: From Boolean to Coupled Reward Machines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Aneta Vulgarakis Feljan, Athanasios Karapantelakis, Jendrik Seipp, Kristina Levina, Nikolaos Pappas","submitted_at":"2025-10-31T10:00:57Z","abstract_excerpt":"Reward machines (RMs) inform reinforcement learning agents about the reward structure of the environment, enabling support for non-Markovian tasks and improving sample efficiency. However, learning with RMs is ill-suited for long-horizon problems where subtasks can be completed in any order. In such cases, the amount of information to learn increases exponentially with the number of unordered subtasks. We address this issue by introducing three generalisations of RMs: (1) Numeric RMs allow users to express complex tasks in a compact form. (2) In agenda RMs, states are associated with an agenda"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.27329","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-10-31T10:00:57Z","cross_cats_sorted":[],"title_canon_sha256":"9609f53f2d9b9f16152fe6c4b286819e183008c62070266dbd6687eabc1b421d","abstract_canon_sha256":"9659df906e5e3bb7559614895e83fd4b8c7a07a3970c6ce084bc50284d896bf0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T02:13:18.006882Z","signature_b64":"5nwfF6P7iU2akSEZaiHp7buqZezgSeFYkIkYmD0ZM2Y8dhqAljmeYvnylSpm0HuS1Lb8iBEnJVutMMxU6MJVAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e1a9095e802ec0b1370f4899f9ae363d189c4b59f05f49edafa83fbc1c7929c","last_reissued_at":"2026-06-23T02:13:18.006351Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T02:13:18.006351Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning for Long-Horizon Unordered Tasks: From Boolean to Coupled Reward Machines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Aneta Vulgarakis Feljan, Athanasios Karapantelakis, Jendrik Seipp, Kristina Levina, Nikolaos Pappas","submitted_at":"2025-10-31T10:00:57Z","abstract_excerpt":"Reward machines (RMs) inform reinforcement learning agents about the reward structure of the environment, enabling support for non-Markovian tasks and improving sample efficiency. However, learning with RMs is ill-suited for long-horizon problems where subtasks can be completed in any order. In such cases, the amount of information to learn increases exponentially with the number of unordered subtasks. We address this issue by introducing three generalisations of RMs: (1) Numeric RMs allow users to express complex tasks in a compact form. (2) In agenda RMs, states are associated with an agenda"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.27329","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.27329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.27329","created_at":"2026-06-23T02:13:18.006420+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.27329v2","created_at":"2026-06-23T02:13:18.006420+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.27329","created_at":"2026-06-23T02:13:18.006420+00:00"},{"alias_kind":"pith_short_12","alias_value":"JYNJBFPIALWA","created_at":"2026-06-23T02:13:18.006420+00:00"},{"alias_kind":"pith_short_16","alias_value":"JYNJBFPIALWAWE3Q","created_at":"2026-06-23T02:13:18.006420+00:00"},{"alias_kind":"pith_short_8","alias_value":"JYNJBFPI","created_at":"2026-06-23T02:13:18.006420+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP","json":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP.json","graph_json":"https://pith.science/api/pith-number/JYNJBFPIALWAWE3Q6SEZ7GXDMP/graph.json","events_json":"https://pith.science/api/pith-number/JYNJBFPIALWAWE3Q6SEZ7GXDMP/events.json","paper":"https://pith.science/paper/JYNJBFPI"},"agent_actions":{"view_html":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP","download_json":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP.json","view_paper":"https://pith.science/paper/JYNJBFPI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.27329&json=true","fetch_graph":"https://pith.science/api/pith-number/JYNJBFPIALWAWE3Q6SEZ7GXDMP/graph.json","fetch_events":"https://pith.science/api/pith-number/JYNJBFPIALWAWE3Q6SEZ7GXDMP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP/action/storage_attestation","attest_author":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP/action/author_attestation","sign_citation":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP/action/citation_signature","submit_replication":"https://pith.science/pith/JYNJBFPIALWAWE3Q6SEZ7GXDMP/action/replication_record"}},"created_at":"2026-06-23T02:13:18.006420+00:00","updated_at":"2026-06-23T02:13:18.006420+00:00"}