{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TWFW3SZEMA3TYYVKXCF2XK262U","short_pith_number":"pith:TWFW3SZE","schema_version":"1.0","canonical_sha256":"9d8b6dcb2460373c62aab88babab5ed53ab640b33faafc77b0aeb4a6dd80d619","source":{"kind":"arxiv","id":"2403.18765","version":1},"attestation_state":"computed","paper":{"title":"CaT: Constraints as Terminations for Legged Locomotion Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Elliot Chane-Sane, Nicolas Mansard, Olivier Stasse, Philippe Sou\\`eres, Pierre-Alexandre Leziart, Thomas Flayols","submitted_at":"2024-03-27T17:03:31Z","abstract_excerpt":"Deep Reinforcement Learning (RL) has demonstrated impressive results in solving complex robotic tasks such as quadruped locomotion. Yet, current solvers fail to produce efficient policies respecting hard constraints. In this work, we advocate for integrating constraints into robot learning and present Constraints as Terminations (CaT), a novel constrained RL algorithm. Departing from classical constrained RL formulations, we reformulate constraints through stochastic terminations during policy learning: any violation of a constraint triggers a probability of terminating potential future reward"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.18765","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-03-27T17:03:31Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"738a8a03707883ca72cc82dc2624c89588b74bff70d0a742860d14858031e609","abstract_canon_sha256":"15543b739fc4322c2eea6d2b2843f475269953b0f747551b8a296dcf34cb5bea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:01:26.869845Z","signature_b64":"dA9NeUaVx6i13IgDLUmYsf+FF8K2s7KPIVZwBj9W9rbn2oh73v0CzkagOQ0CnScNWAdyPbwVq7TCliGRI/SqBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d8b6dcb2460373c62aab88babab5ed53ab640b33faafc77b0aeb4a6dd80d619","last_reissued_at":"2026-07-05T08:01:26.869414Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:01:26.869414Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CaT: Constraints as Terminations for Legged Locomotion Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Elliot Chane-Sane, Nicolas Mansard, Olivier Stasse, Philippe Sou\\`eres, Pierre-Alexandre Leziart, Thomas Flayols","submitted_at":"2024-03-27T17:03:31Z","abstract_excerpt":"Deep Reinforcement Learning (RL) has demonstrated impressive results in solving complex robotic tasks such as quadruped locomotion. Yet, current solvers fail to produce efficient policies respecting hard constraints. In this work, we advocate for integrating constraints into robot learning and present Constraints as Terminations (CaT), a novel constrained RL algorithm. Departing from classical constrained RL formulations, we reformulate constraints through stochastic terminations during policy learning: any violation of a constraint triggers a probability of terminating potential future reward"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.18765","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.18765/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.18765","created_at":"2026-07-05T08:01:26.869463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.18765v1","created_at":"2026-07-05T08:01:26.869463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.18765","created_at":"2026-07-05T08:01:26.869463+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWFW3SZEMA3T","created_at":"2026-07-05T08:01:26.869463+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWFW3SZEMA3TYYVK","created_at":"2026-07-05T08:01:26.869463+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWFW3SZE","created_at":"2026-07-05T08:01:26.869463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00215","citing_title":"First Order Model-Based RL through Decoupled Backpropagation","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U","json":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U.json","graph_json":"https://pith.science/api/pith-number/TWFW3SZEMA3TYYVKXCF2XK262U/graph.json","events_json":"https://pith.science/api/pith-number/TWFW3SZEMA3TYYVKXCF2XK262U/events.json","paper":"https://pith.science/paper/TWFW3SZE"},"agent_actions":{"view_html":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U","download_json":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U.json","view_paper":"https://pith.science/paper/TWFW3SZE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.18765&json=true","fetch_graph":"https://pith.science/api/pith-number/TWFW3SZEMA3TYYVKXCF2XK262U/graph.json","fetch_events":"https://pith.science/api/pith-number/TWFW3SZEMA3TYYVKXCF2XK262U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U/action/storage_attestation","attest_author":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U/action/author_attestation","sign_citation":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U/action/citation_signature","submit_replication":"https://pith.science/pith/TWFW3SZEMA3TYYVKXCF2XK262U/action/replication_record"}},"created_at":"2026-07-05T08:01:26.869463+00:00","updated_at":"2026-07-05T08:01:26.869463+00:00"}