{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:COC5TPBHO5DWKJBNRGUNWT73C5","short_pith_number":"pith:COC5TPBH","schema_version":"1.0","canonical_sha256":"1385d9bc27774765242d89a8db4ffb174481021acc6357a29bda2dd26bb71ea1","source":{"kind":"arxiv","id":"2410.23208","version":2},"attestation_state":"computed","paper":{"title":"Kinetix: Investigating the Training of General Agents through Open-Ended Physics-Based Control Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chris Lu, Jakob Foerster, Michael Beukman, Michael Matthews","submitted_at":"2024-10-30T16:59:41Z","abstract_excerpt":"While large models trained with self-supervised learning on offline datasets have shown remarkable capabilities in text and image domains, achieving the same generalisation for agents that act in sequential decision problems remains an open challenge. In this work, we take a step towards this goal by procedurally generating tens of millions of 2D physics-based tasks and using these to train a general reinforcement learning (RL) agent for physical control. To this end, we introduce Kinetix: an open-ended space of physics-based RL environments that can represent tasks ranging from robotic locomo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.23208","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-30T16:59:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b54ad1848c4f98787ad4f58c6c0554ead140bdc10832007072c41f1bd59f63eb","abstract_canon_sha256":"2751945e94d699ff53dd7ae11ed24b8f25f759e27c3cacb3513bb1cd14dd7a15"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:53.220447Z","signature_b64":"tjiyqmkYRXM9gGxhtye3vF/MqZmYGvX4X38d3C/Nya/dueIIIQfoBYjworLLHSEK0TuWm42mzwKRhbNutY2kAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1385d9bc27774765242d89a8db4ffb174481021acc6357a29bda2dd26bb71ea1","last_reissued_at":"2026-07-05T10:22:53.219607Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:53.219607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Kinetix: Investigating the Training of General Agents through Open-Ended Physics-Based Control Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chris Lu, Jakob Foerster, Michael Beukman, Michael Matthews","submitted_at":"2024-10-30T16:59:41Z","abstract_excerpt":"While large models trained with self-supervised learning on offline datasets have shown remarkable capabilities in text and image domains, achieving the same generalisation for agents that act in sequential decision problems remains an open challenge. In this work, we take a step towards this goal by procedurally generating tens of millions of 2D physics-based tasks and using these to train a general reinforcement learning (RL) agent for physical control. To this end, we introduce Kinetix: an open-ended space of physics-based RL environments that can represent tasks ranging from robotic locomo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.23208","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.23208/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.23208","created_at":"2026-07-05T10:22:53.219708+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.23208v2","created_at":"2026-07-05T10:22:53.219708+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.23208","created_at":"2026-07-05T10:22:53.219708+00:00"},{"alias_kind":"pith_short_12","alias_value":"COC5TPBHO5DW","created_at":"2026-07-05T10:22:53.219708+00:00"},{"alias_kind":"pith_short_16","alias_value":"COC5TPBHO5DWKJBN","created_at":"2026-07-05T10:22:53.219708+00:00"},{"alias_kind":"pith_short_8","alias_value":"COC5TPBH","created_at":"2026-07-05T10:22:53.219708+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28529","citing_title":"The Speedup Paradox: Rethinking Inference Speed-Quality Trade-off in Embodied Tasks","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28529","citing_title":"The Speedup Paradox: Rethinking Inference Speed-Quality Trade-off in Embodied Tasks","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25050","citing_title":"DiscreteRTC: Discrete Diffusion Policies are Natural Asynchronous Executors","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25537","citing_title":"Action-Prior Denoising for Smooth Real-Time Chunking","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01665","citing_title":"TABX: A High-Throughput Sandbox Battle Simulator for Multi-Agent Reinforcement Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23551","citing_title":"Goal-Conditioned Agents that Learn Everything All at Once","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07339","citing_title":"Real-Time Execution of Action Chunking Flow Policies","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08168","citing_title":"Understanding Asynchronous Inference Methods for Vision-Language-Action Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25050","citing_title":"DiscreteRTC: Discrete Diffusion Policies are Natural Asynchronous Executors","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5","json":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5.json","graph_json":"https://pith.science/api/pith-number/COC5TPBHO5DWKJBNRGUNWT73C5/graph.json","events_json":"https://pith.science/api/pith-number/COC5TPBHO5DWKJBNRGUNWT73C5/events.json","paper":"https://pith.science/paper/COC5TPBH"},"agent_actions":{"view_html":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5","download_json":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5.json","view_paper":"https://pith.science/paper/COC5TPBH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.23208&json=true","fetch_graph":"https://pith.science/api/pith-number/COC5TPBHO5DWKJBNRGUNWT73C5/graph.json","fetch_events":"https://pith.science/api/pith-number/COC5TPBHO5DWKJBNRGUNWT73C5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5/action/storage_attestation","attest_author":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5/action/author_attestation","sign_citation":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5/action/citation_signature","submit_replication":"https://pith.science/pith/COC5TPBHO5DWKJBNRGUNWT73C5/action/replication_record"}},"created_at":"2026-07-05T10:22:53.219708+00:00","updated_at":"2026-07-05T10:22:53.219708+00:00"}