{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6NGSCF2BCWNVY5NE6ZMPAFUDLM","short_pith_number":"pith:6NGSCF2B","schema_version":"1.0","canonical_sha256":"f34d211741159b5c75a4f658f016835b24e2b8aca69a2ab8e9fb058fc8db3abf","source":{"kind":"arxiv","id":"2312.03126","version":2},"attestation_state":"computed","paper":{"title":"Learning Curricula in Open-Ended Worlds","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Minqi Jiang","submitted_at":"2023-12-03T16:44:00Z","abstract_excerpt":"Deep reinforcement learning (RL) provides powerful methods for training optimal sequential decision-making agents. As collecting real-world interactions can entail additional costs and safety risks, the common paradigm of sim2real conducts training in a simulator, followed by real-world deployment. Unfortunately, RL agents easily overfit to the choice of simulated training environments, and worse still, learning ends when the agent masters the specific set of simulated environments. In contrast, the real world is highly open-ended, featuring endlessly evolving environments and challenges, maki"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.03126","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-03T16:44:00Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8a4306877117052084b01b484d2dad228a6f6da1cc790f094d5fb52391b4bc7d","abstract_canon_sha256":"b743f68f1b436f7ae5b86d9f65b8438173f9e84f076cf5c374723e4c0c2d37a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:21:47.122081Z","signature_b64":"X1VIiBbCTTJA2rIed4QLdSZoAdF6xjF5Tr++JgIa6XR+3gGgEjJB2c2zl+UR090wJQpUQb88LpxIVNZnlXiOBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f34d211741159b5c75a4f658f016835b24e2b8aca69a2ab8e9fb058fc8db3abf","last_reissued_at":"2026-07-05T07:21:47.121410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:21:47.121410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Curricula in Open-Ended Worlds","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Minqi Jiang","submitted_at":"2023-12-03T16:44:00Z","abstract_excerpt":"Deep reinforcement learning (RL) provides powerful methods for training optimal sequential decision-making agents. As collecting real-world interactions can entail additional costs and safety risks, the common paradigm of sim2real conducts training in a simulator, followed by real-world deployment. Unfortunately, RL agents easily overfit to the choice of simulated training environments, and worse still, learning ends when the agent masters the specific set of simulated environments. In contrast, the real world is highly open-ended, featuring endlessly evolving environments and challenges, maki"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03126","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.03126/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.03126","created_at":"2026-07-05T07:21:47.121481+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.03126v2","created_at":"2026-07-05T07:21:47.121481+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03126","created_at":"2026-07-05T07:21:47.121481+00:00"},{"alias_kind":"pith_short_12","alias_value":"6NGSCF2BCWNV","created_at":"2026-07-05T07:21:47.121481+00:00"},{"alias_kind":"pith_short_16","alias_value":"6NGSCF2BCWNVY5NE","created_at":"2026-07-05T07:21:47.121481+00:00"},{"alias_kind":"pith_short_8","alias_value":"6NGSCF2B","created_at":"2026-07-05T07:21:47.121481+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08193","citing_title":"Open-ended Multi-agent Autocurricula via Visual Inspection of Policies with Multi-modal LLMs","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM","json":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM.json","graph_json":"https://pith.science/api/pith-number/6NGSCF2BCWNVY5NE6ZMPAFUDLM/graph.json","events_json":"https://pith.science/api/pith-number/6NGSCF2BCWNVY5NE6ZMPAFUDLM/events.json","paper":"https://pith.science/paper/6NGSCF2B"},"agent_actions":{"view_html":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM","download_json":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM.json","view_paper":"https://pith.science/paper/6NGSCF2B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.03126&json=true","fetch_graph":"https://pith.science/api/pith-number/6NGSCF2BCWNVY5NE6ZMPAFUDLM/graph.json","fetch_events":"https://pith.science/api/pith-number/6NGSCF2BCWNVY5NE6ZMPAFUDLM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM/action/storage_attestation","attest_author":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM/action/author_attestation","sign_citation":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM/action/citation_signature","submit_replication":"https://pith.science/pith/6NGSCF2BCWNVY5NE6ZMPAFUDLM/action/replication_record"}},"created_at":"2026-07-05T07:21:47.121481+00:00","updated_at":"2026-07-05T07:21:47.121481+00:00"}