{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:KS5KKHWAQGQ3HAH45ZAO2BQ3F7","short_pith_number":"pith:KS5KKHWA","schema_version":"1.0","canonical_sha256":"54baa51ec081a1b380fcee40ed061b2fe5e6023303c3bb631f3aae6c447c2b24","source":{"kind":"arxiv","id":"2212.14530","version":1},"attestation_state":"computed","paper":{"title":"POMRL: No-Regret Learning-to-Plan with Increasing Horizons","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Brendan O'Donoghue, Claire Vernade, Khimya Khetarpal, Satinder Singh, Tom Zahavy","submitted_at":"2022-12-30T03:09:45Z","abstract_excerpt":"We study the problem of planning under model uncertainty in an online meta-reinforcement learning (RL) setting where an agent is presented with a sequence of related tasks with limited interactions per task. The agent can use its experience in each task and across tasks to estimate both the transition model and the distribution over tasks. We propose an algorithm to meta-learn the underlying structure across tasks, utilize it to plan in each task, and upper-bound the regret of the planning loss. Our bound suggests that the average regret over tasks decreases as the number of tasks increases an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.14530","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2022-12-30T03:09:45Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"0d01a289659349b95aa1b7f18b2c203f39874580136bb7916e8a93e31a13312f","abstract_canon_sha256":"155c7daa63bb94a52d9584161a26311924201ea0386286e2c7bc20c569f29ad7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:29:11.327450Z","signature_b64":"DxA+paTufmIm0ThJvhN/UUlV9B0XSsE88yAhPDIbu+4/o87x2LKAN5xiRHvVox8OeX/yxgai3/+gKfinjF5gBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54baa51ec081a1b380fcee40ed061b2fe5e6023303c3bb631f3aae6c447c2b24","last_reissued_at":"2026-07-05T05:29:11.326878Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:29:11.326878Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"POMRL: No-Regret Learning-to-Plan with Increasing Horizons","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Brendan O'Donoghue, Claire Vernade, Khimya Khetarpal, Satinder Singh, Tom Zahavy","submitted_at":"2022-12-30T03:09:45Z","abstract_excerpt":"We study the problem of planning under model uncertainty in an online meta-reinforcement learning (RL) setting where an agent is presented with a sequence of related tasks with limited interactions per task. The agent can use its experience in each task and across tasks to estimate both the transition model and the distribution over tasks. We propose an algorithm to meta-learn the underlying structure across tasks, utilize it to plan in each task, and upper-bound the regret of the planning loss. Our bound suggests that the average regret over tasks decreases as the number of tasks increases an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.14530","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.14530/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.14530","created_at":"2026-07-05T05:29:11.326942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.14530v1","created_at":"2026-07-05T05:29:11.326942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.14530","created_at":"2026-07-05T05:29:11.326942+00:00"},{"alias_kind":"pith_short_12","alias_value":"KS5KKHWAQGQ3","created_at":"2026-07-05T05:29:11.326942+00:00"},{"alias_kind":"pith_short_16","alias_value":"KS5KKHWAQGQ3HAH4","created_at":"2026-07-05T05:29:11.326942+00:00"},{"alias_kind":"pith_short_8","alias_value":"KS5KKHWA","created_at":"2026-07-05T05:29:11.326942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7","json":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7.json","graph_json":"https://pith.science/api/pith-number/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/graph.json","events_json":"https://pith.science/api/pith-number/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/events.json","paper":"https://pith.science/paper/KS5KKHWA"},"agent_actions":{"view_html":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7","download_json":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7.json","view_paper":"https://pith.science/paper/KS5KKHWA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.14530&json=true","fetch_graph":"https://pith.science/api/pith-number/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/graph.json","fetch_events":"https://pith.science/api/pith-number/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/action/storage_attestation","attest_author":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/action/author_attestation","sign_citation":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/action/citation_signature","submit_replication":"https://pith.science/pith/KS5KKHWAQGQ3HAH45ZAO2BQ3F7/action/replication_record"}},"created_at":"2026-07-05T05:29:11.326942+00:00","updated_at":"2026-07-05T05:29:11.326942+00:00"}