{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YNLFJIRPQ5VRHVKT6F3SCIGVD2","short_pith_number":"pith:YNLFJIRP","schema_version":"1.0","canonical_sha256":"c35654a22f876b13d553f1772120d51e9c4123731a76457b080ea00275a47a8b","source":{"kind":"arxiv","id":"2305.04412","version":1},"attestation_state":"computed","paper":{"title":"Efficient Reinforcement Learning for Autonomous Driving with Parameterized Skills and Priors","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Hao Shao, Jie Liu, Letian Wang, Ruobing Chen, Steven L. Waslander, Wenshuo Wang, Yu Liu","submitted_at":"2023-05-08T01:39:35Z","abstract_excerpt":"When autonomous vehicles are deployed on public roads, they will encounter countless and diverse driving situations. Many manually designed driving policies are difficult to scale to the real world. Fortunately, reinforcement learning has shown great success in many tasks by automatic trial and error. However, when it comes to autonomous driving in interactive dense traffic, RL agents either fail to learn reasonable performance or necessitate a large amount of data. Our insight is that when humans learn to drive, they will 1) make decisions over the high-level skill space instead of the low-le"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.04412","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2023-05-08T01:39:35Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"fe476ed657af7be4724ba4456801c1fdfce42abef150b1af03ef71dc3c829120","abstract_canon_sha256":"a3ca6ac3917d8f5ba72da4c6603a4e45b789ee3180ad8e9afa2a52b295a55c8f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:07:49.765143Z","signature_b64":"jbmXWGf2u+9q8td4nQGqkcezVag9Npr5iilzFXADbbRI9loORGHtotKQ5x8/c3iWVdiVh6H9l0D//qmgtjYjBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c35654a22f876b13d553f1772120d51e9c4123731a76457b080ea00275a47a8b","last_reissued_at":"2026-07-05T06:07:49.764782Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:07:49.764782Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Reinforcement Learning for Autonomous Driving with Parameterized Skills and Priors","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Hao Shao, Jie Liu, Letian Wang, Ruobing Chen, Steven L. Waslander, Wenshuo Wang, Yu Liu","submitted_at":"2023-05-08T01:39:35Z","abstract_excerpt":"When autonomous vehicles are deployed on public roads, they will encounter countless and diverse driving situations. Many manually designed driving policies are difficult to scale to the real world. Fortunately, reinforcement learning has shown great success in many tasks by automatic trial and error. However, when it comes to autonomous driving in interactive dense traffic, RL agents either fail to learn reasonable performance or necessitate a large amount of data. Our insight is that when humans learn to drive, they will 1) make decisions over the high-level skill space instead of the low-le"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.04412","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.04412/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.04412","created_at":"2026-07-05T06:07:49.764841+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.04412v1","created_at":"2026-07-05T06:07:49.764841+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.04412","created_at":"2026-07-05T06:07:49.764841+00:00"},{"alias_kind":"pith_short_12","alias_value":"YNLFJIRPQ5VR","created_at":"2026-07-05T06:07:49.764841+00:00"},{"alias_kind":"pith_short_16","alias_value":"YNLFJIRPQ5VRHVKT","created_at":"2026-07-05T06:07:49.764841+00:00"},{"alias_kind":"pith_short_8","alias_value":"YNLFJIRP","created_at":"2026-07-05T06:07:49.764841+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20641","citing_title":"MAGNIFIED: RL Fine-tuning of Multimodal Large Language Models for Motion Planning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12224","citing_title":"Intrinsic Vicarious Conditioning for Deep Reinforcement Learning","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2","json":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2.json","graph_json":"https://pith.science/api/pith-number/YNLFJIRPQ5VRHVKT6F3SCIGVD2/graph.json","events_json":"https://pith.science/api/pith-number/YNLFJIRPQ5VRHVKT6F3SCIGVD2/events.json","paper":"https://pith.science/paper/YNLFJIRP"},"agent_actions":{"view_html":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2","download_json":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2.json","view_paper":"https://pith.science/paper/YNLFJIRP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.04412&json=true","fetch_graph":"https://pith.science/api/pith-number/YNLFJIRPQ5VRHVKT6F3SCIGVD2/graph.json","fetch_events":"https://pith.science/api/pith-number/YNLFJIRPQ5VRHVKT6F3SCIGVD2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2/action/storage_attestation","attest_author":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2/action/author_attestation","sign_citation":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2/action/citation_signature","submit_replication":"https://pith.science/pith/YNLFJIRPQ5VRHVKT6F3SCIGVD2/action/replication_record"}},"created_at":"2026-07-05T06:07:49.764841+00:00","updated_at":"2026-07-05T06:07:49.764841+00:00"}