{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:XK2QDROZYBZBDYQFBOUZHKKZOC","short_pith_number":"pith:XK2QDROZ","canonical_record":{"source":{"id":"2504.17838","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-24T17:56:01Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"7a17e2b8eb3398e51200e8d5e6bce1bf42ad2ebec26157fc9dbf2cfdb92523fb","abstract_canon_sha256":"74a764977ecb121428eff0c4c2ce5cc9edbfaa052d7ec7c0bebdf788a05806ad"},"schema_version":"1.0"},"canonical_sha256":"bab501c5d9c07211e2050ba993a9597090ee54984c5924572e4a9d5c0b8a450d","source":{"kind":"arxiv","id":"2504.17838","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.17838","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"arxiv_version","alias_value":"2504.17838v3","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.17838","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_12","alias_value":"XK2QDROZYBZB","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_16","alias_value":"XK2QDROZYBZBDYQF","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_8","alias_value":"XK2QDROZ","created_at":"2026-07-05T11:56:56Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:XK2QDROZYBZBDYQFBOUZHKKZOC","target":"record","payload":{"canonical_record":{"source":{"id":"2504.17838","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-24T17:56:01Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"7a17e2b8eb3398e51200e8d5e6bce1bf42ad2ebec26157fc9dbf2cfdb92523fb","abstract_canon_sha256":"74a764977ecb121428eff0c4c2ce5cc9edbfaa052d7ec7c0bebdf788a05806ad"},"schema_version":"1.0"},"canonical_sha256":"bab501c5d9c07211e2050ba993a9597090ee54984c5924572e4a9d5c0b8a450d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:56:56.470517Z","signature_b64":"l6w1keAkCFQGzEOo4zKp+nLg2FwtV5MsCaYfpxuHAHLlfrejYLd4LzGNHCJzuVjrd35RwedPukboi4Sc7qvMCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bab501c5d9c07211e2050ba993a9597090ee54984c5924572e4a9d5c0b8a450d","last_reissued_at":"2026-07-05T11:56:56.469958Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:56:56.469958Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2504.17838","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:56:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QK5uZmJB+n+YBW4cLBqhZPDv5WYdZz8diiilnBuuF8NUXTQH8V2CXZ5zYD6OWWUywqWf1eK69+28CHriFoB8Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T10:04:42.005654Z"},"content_sha256":"7df9d0bdf00076304341a5904957547cc555721bc54830d3ee99e3ad4ea42751","schema_version":"1.0","event_id":"sha256:7df9d0bdf00076304341a5904957547cc555721bc54830d3ee99e3ad4ea42751"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:XK2QDROZYBZBDYQFBOUZHKKZOC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"CaRL: Learning Scalable Planning Policies with Simple Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Andreas Geiger, Bernhard Jaeger, Daniel Dauner, Jens Bei{\\ss}wenger, Kashyap Chitta, Simon Gerstenecker","submitted_at":"2025-04-24T17:56:01Z","abstract_excerpt":"We investigate reinforcement learning (RL) for privileged planning in autonomous driving. State-of-the-art approaches for this task are rule-based, but these methods do not scale to the long tail. RL, on the other hand, is scalable and does not suffer from compounding errors like imitation learning. Contemporary RL approaches for driving use complex shaped rewards that sum multiple individual rewards, \\eg~progress, position, or orientation rewards. We show that PPO fails to optimize a popular version of these rewards when the mini-batch size is increased, which limits the scalability of these "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.17838","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.17838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:56:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aWxxqmmgSokyTQEcjt4WsAFFd81erAFI/6RID0T85J8q83SsHHpVkqXX+LqMr27Zldx21wufrXwlWZsapRLhCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T10:04:42.006579Z"},"content_sha256":"15faf1f2cb52f5fa9bb2ae747cfe1c8409fddeb4e8de851896c7e3cd7dd53802","schema_version":"1.0","event_id":"sha256:15faf1f2cb52f5fa9bb2ae747cfe1c8409fddeb4e8de851896c7e3cd7dd53802"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/bundle.json","state_url":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T10:04:42Z","links":{"resolver":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC","bundle":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/bundle.json","state":"https://pith.science/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XK2QDROZYBZBDYQFBOUZHKKZOC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:XK2QDROZYBZBDYQFBOUZHKKZOC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"74a764977ecb121428eff0c4c2ce5cc9edbfaa052d7ec7c0bebdf788a05806ad","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-24T17:56:01Z","title_canon_sha256":"7a17e2b8eb3398e51200e8d5e6bce1bf42ad2ebec26157fc9dbf2cfdb92523fb"},"schema_version":"1.0","source":{"id":"2504.17838","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.17838","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"arxiv_version","alias_value":"2504.17838v3","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.17838","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_12","alias_value":"XK2QDROZYBZB","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_16","alias_value":"XK2QDROZYBZBDYQF","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_8","alias_value":"XK2QDROZ","created_at":"2026-07-05T11:56:56Z"}],"graph_snapshots":[{"event_id":"sha256:15faf1f2cb52f5fa9bb2ae747cfe1c8409fddeb4e8de851896c7e3cd7dd53802","target":"graph","created_at":"2026-07-05T11:56:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.17838/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We investigate reinforcement learning (RL) for privileged planning in autonomous driving. State-of-the-art approaches for this task are rule-based, but these methods do not scale to the long tail. RL, on the other hand, is scalable and does not suffer from compounding errors like imitation learning. Contemporary RL approaches for driving use complex shaped rewards that sum multiple individual rewards, \\eg~progress, position, or orientation rewards. We show that PPO fails to optimize a popular version of these rewards when the mini-batch size is increased, which limits the scalability of these ","authors_text":"Andreas Geiger, Bernhard Jaeger, Daniel Dauner, Jens Bei{\\ss}wenger, Kashyap Chitta, Simon Gerstenecker","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-24T17:56:01Z","title":"CaRL: Learning Scalable Planning Policies with Simple Rewards"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.17838","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:7df9d0bdf00076304341a5904957547cc555721bc54830d3ee99e3ad4ea42751","target":"record","created_at":"2026-07-05T11:56:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"74a764977ecb121428eff0c4c2ce5cc9edbfaa052d7ec7c0bebdf788a05806ad","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-24T17:56:01Z","title_canon_sha256":"7a17e2b8eb3398e51200e8d5e6bce1bf42ad2ebec26157fc9dbf2cfdb92523fb"},"schema_version":"1.0","source":{"id":"2504.17838","kind":"arxiv","version":3}},"canonical_sha256":"bab501c5d9c07211e2050ba993a9597090ee54984c5924572e4a9d5c0b8a450d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"bab501c5d9c07211e2050ba993a9597090ee54984c5924572e4a9d5c0b8a450d","first_computed_at":"2026-07-05T11:56:56.469958Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:56:56.469958Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"l6w1keAkCFQGzEOo4zKp+nLg2FwtV5MsCaYfpxuHAHLlfrejYLd4LzGNHCJzuVjrd35RwedPukboi4Sc7qvMCw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:56:56.470517Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.17838","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:7df9d0bdf00076304341a5904957547cc555721bc54830d3ee99e3ad4ea42751","sha256:15faf1f2cb52f5fa9bb2ae747cfe1c8409fddeb4e8de851896c7e3cd7dd53802"],"state_sha256":"6e7d36c30458a67e66f1ab78eae42d07e1fb14262c60a498c62b9033eb31116e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"69cneTZfdwKSSISJx5NGSDOIaXKoxVUcGplPw0NkzLy23u1EzL3aIZwh90Cu7ouSRgavAscFZpvXLHk9S/L7BQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T10:04:42.012812Z","bundle_sha256":"97b081e17e3bd9fb0b6ab56c98f756fbd2b1dd93bc459086676ad79bfeb5a907"}}