{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:DAKDXFIMRSJG67P2ASOGGMPOYN","short_pith_number":"pith:DAKDXFIM","schema_version":"1.0","canonical_sha256":"18143b950c8c926f7dfa049c6331eec3443041a2888f4724b31ead7132fdc086","source":{"kind":"arxiv","id":"2010.10560","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning for Optimization of COVID-19 Mitigation policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.LG","authors_text":"Jonathan Browne, Lauren Meyers, Peter Stone, Peter Wurman, Roberto Capobianco, Spencer Fox, Stacy Jong, Varun Kompella","submitted_at":"2020-10-20T18:40:15Z","abstract_excerpt":"The year 2020 has seen the COVID-19 virus lead to one of the worst global pandemics in history. As a result, governments around the world are faced with the challenge of protecting public health, while keeping the economy running to the greatest extent possible. Epidemiological models provide insight into the spread of these types of diseases and predict the effects of possible intervention policies. However, to date,the even the most data-driven intervention policies rely on heuristics. In this paper, we study how reinforcement learning (RL) can be used to optimize mitigation policies that mi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.10560","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-20T18:40:15Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"4924aeef15b51cf23a2681e1b8c880f84cf659509b04728b1ee6b0f0cbbed463","abstract_canon_sha256":"9f710f029e70f0fed5de2737605caf9418db06d54b72ee19f215787342f25978"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:44:52.372169Z","signature_b64":"X3Id/aZkjW9qOPHwg2Yywewm26I0+F722qXTn6dFOAbaqLpuRCS2UGWhYjgEwowc6IGHQmo1qPiT0E1Me+t5DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18143b950c8c926f7dfa049c6331eec3443041a2888f4724b31ead7132fdc086","last_reissued_at":"2026-07-05T01:44:52.371718Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:44:52.371718Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning for Optimization of COVID-19 Mitigation policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.LG","authors_text":"Jonathan Browne, Lauren Meyers, Peter Stone, Peter Wurman, Roberto Capobianco, Spencer Fox, Stacy Jong, Varun Kompella","submitted_at":"2020-10-20T18:40:15Z","abstract_excerpt":"The year 2020 has seen the COVID-19 virus lead to one of the worst global pandemics in history. As a result, governments around the world are faced with the challenge of protecting public health, while keeping the economy running to the greatest extent possible. Epidemiological models provide insight into the spread of these types of diseases and predict the effects of possible intervention policies. However, to date,the even the most data-driven intervention policies rely on heuristics. In this paper, we study how reinforcement learning (RL) can be used to optimize mitigation policies that mi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.10560","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.10560/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.10560","created_at":"2026-07-05T01:44:52.371782+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.10560v1","created_at":"2026-07-05T01:44:52.371782+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.10560","created_at":"2026-07-05T01:44:52.371782+00:00"},{"alias_kind":"pith_short_12","alias_value":"DAKDXFIMRSJG","created_at":"2026-07-05T01:44:52.371782+00:00"},{"alias_kind":"pith_short_16","alias_value":"DAKDXFIMRSJG67P2","created_at":"2026-07-05T01:44:52.371782+00:00"},{"alias_kind":"pith_short_8","alias_value":"DAKDXFIM","created_at":"2026-07-05T01:44:52.371782+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04562","citing_title":"Neetyabhas: A Framework for Uncertainty-Aware Public Policy Optimization in Rational Agent-Based Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2311.00855","citing_title":"A Multi-Agent Reinforcement Learning Framework for Public Health Decision Analysis","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19397","citing_title":"Optimizing Resource-Constrained Non-Pharmaceutical Interventions for Multi-Cluster Outbreak Control Using Hierarchical Reinforcement Learning","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN","json":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN.json","graph_json":"https://pith.science/api/pith-number/DAKDXFIMRSJG67P2ASOGGMPOYN/graph.json","events_json":"https://pith.science/api/pith-number/DAKDXFIMRSJG67P2ASOGGMPOYN/events.json","paper":"https://pith.science/paper/DAKDXFIM"},"agent_actions":{"view_html":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN","download_json":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN.json","view_paper":"https://pith.science/paper/DAKDXFIM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.10560&json=true","fetch_graph":"https://pith.science/api/pith-number/DAKDXFIMRSJG67P2ASOGGMPOYN/graph.json","fetch_events":"https://pith.science/api/pith-number/DAKDXFIMRSJG67P2ASOGGMPOYN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN/action/storage_attestation","attest_author":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN/action/author_attestation","sign_citation":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN/action/citation_signature","submit_replication":"https://pith.science/pith/DAKDXFIMRSJG67P2ASOGGMPOYN/action/replication_record"}},"created_at":"2026-07-05T01:44:52.371782+00:00","updated_at":"2026-07-05T01:44:52.371782+00:00"}