{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XK2QDROZYBZBDYQFBOUZHKKZOC","short_pith_number":"pith:XK2QDROZ","schema_version":"1.0","canonical_sha256":"bab501c5d9c07211e2050ba993a9597090ee54984c5924572e4a9d5c0b8a450d","source":{"kind":"arxiv","id":"2504.17838","version":3},"attestation_state":"computed","paper":{"title":"CaRL: Learning Scalable Planning Policies with Simple Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Andreas Geiger, Bernhard Jaeger, Daniel Dauner, Jens Bei{\\ss}wenger, Kashyap Chitta, Simon Gerstenecker","submitted_at":"2025-04-24T17:56:01Z","abstract_excerpt":"We investigate reinforcement learning (RL) for privileged planning in autonomous driving. State-of-the-art approaches for this task are rule-based, but these methods do not scale to the long tail. RL, on the other hand, is scalable and does not suffer from compounding errors like imitation learning. Contemporary RL approaches for driving use complex shaped rewards that sum multiple individual rewards, \\eg~progress, position, or orientation rewards. We show that PPO fails to optimize a popular version of these rewards when the mini-batch size is increased, which limits the scalability of these "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.17838","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-24T17:56:01Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"7a17e2b8eb3398e51200e8d5e6bce1bf42ad2ebec26157fc9dbf2cfdb92523fb","abstract_canon_sha256":"74a764977ecb121428eff0c4c2ce5cc9edbfaa052d7ec7c0bebdf788a05806ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:56:56.470517Z","signature_b64":"l6w1keAkCFQGzEOo4zKp+nLg2FwtV5MsCaYfpxuHAHLlfrejYLd4LzGNHCJzuVjrd35RwedPukboi4Sc7qvMCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bab501c5d9c07211e2050ba993a9597090ee54984c5924572e4a9d5c0b8a450d","last_reissued_at":"2026-07-05T11:56:56.469958Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:56:56.469958Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CaRL: Learning Scalable Planning Policies with Simple Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Andreas Geiger, Bernhard Jaeger, Daniel Dauner, Jens Bei{\\ss}wenger, Kashyap Chitta, Simon Gerstenecker","submitted_at":"2025-04-24T17:56:01Z","abstract_excerpt":"We investigate reinforcement learning (RL) for privileged planning in autonomous driving. State-of-the-art approaches for this task are rule-based, but these methods do not scale to the long tail. RL, on the other hand, is scalable and does not suffer from compounding errors like imitation learning. Contemporary RL approaches for driving use complex shaped rewards that sum multiple individual rewards, \\eg~progress, position, or orientation rewards. We show that PPO fails to optimize a popular version of these rewards when the mini-batch size is increased, which limits the scalability of these "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.17838","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.17838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.17838","created_at":"2026-07-05T11:56:56.470019+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.17838v3","created_at":"2026-07-05T11:56:56.470019+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.17838","created_at":"2026-07-05T11:56:56.470019+00:00"},{"alias_kind":"pith_short_12","alias_value":"XK2QDROZYBZB","created_at":"2026-07-05T11:56:56.470019+00:00"},{"alias_kind":"pith_short_16","alias_value":"XK2QDROZYBZBDYQF","created_at":"2026-07-05T11:56:56.470019+00:00"},{"alias_kind":"pith_short_8","alias_value":"XK2QDROZ","created_at":"2026-07-05T11:56:56.470019+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26858","citing_title":"PlanRL: A Trajectory Planning Architecture for Reinforcement Learning-based Driving Experts","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19641","citing_title":"Scaling Self-Play for End-to-End Driving","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14201","citing_title":"MAPLE: Latent Multi-Agent Play for End-to-End Autonomous Driving","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16737","citing_title":"DriveSafer: End-to-End Autonomous Driving with Safety Guidance","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14201","citing_title":"MAPLE: Latent Multi-Agent Play for End-to-End Autonomous Driving","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.24155","citing_title":"Goal-Oriented Reactive Simulation for Closed-Loop Trajectory Prediction","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04138","citing_title":"Learning Dexterous Grasping from Sparse Taxonomy Guidance","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10034","citing_title":"Beyond Self-Play and Scale: A Behavior Benchmark for Generalization in Autonomous Driving","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08535","citing_title":"Fail2Drive: Benchmarking Closed-Loop Driving Generalization","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC","json":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC.json","graph_json":"https://pith.science/api/pith-number/XK2QDROZYBZBDYQFBOUZHKKZOC/graph.json","events_json":"https://pith.science/api/pith-number/XK2QDROZYBZBDYQFBOUZHKKZOC/events.json","paper":"https://pith.science/paper/XK2QDROZ"},"agent_actions":{"view_html":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC","download_json":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC.json","view_paper":"https://pith.science/paper/XK2QDROZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.17838&json=true","fetch_graph":"https://pith.science/api/pith-number/XK2QDROZYBZBDYQFBOUZHKKZOC/graph.json","fetch_events":"https://pith.science/api/pith-number/XK2QDROZYBZBDYQFBOUZHKKZOC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/action/storage_attestation","attest_author":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/action/author_attestation","sign_citation":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/action/citation_signature","submit_replication":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/action/replication_record"}},"created_at":"2026-07-05T11:56:56.470019+00:00","updated_at":"2026-07-05T11:56:56.470019+00:00"}