{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2016:NCLQY5DOKB44DZZOLT5XC6UZF4","short_pith_number":"pith:NCLQY5DO","schema_version":"1.0","canonical_sha256":"68970c746e5079c1e72e5cfb717a992f10a1141b4c3691b65b8c526c5128ca75","source":{"kind":"arxiv","id":"1610.03295","version":1},"attestation_state":"computed","paper":{"title":"Safe, Multi-Agent, Reinforcement Learning for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Amnon Shashua, Shai Shalev-Shwartz, Shaked Shammah","submitted_at":"2016-10-11T12:09:03Z","abstract_excerpt":"Autonomous driving is a multi-agent setting where the host vehicle must apply sophisticated negotiation skills with other road users when overtaking, giving way, merging, taking left and right turns and while pushing ahead in unstructured urban roadways. Since there are many possible scenarios, manually tackling all possible cases will likely yield a too simplistic policy. Moreover, one must balance between unexpected behavior of other drivers/pedestrians and at the same time not to be too defensive so that normal traffic flow is maintained.\n  In this paper we apply deep reinforcement learning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1610.03295","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2016-10-11T12:09:03Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"3c8ae375ea851b2cd293d70077a1c30d1f84f4892bc1983fa8fee06513a10f79","abstract_canon_sha256":"db4f009f01f0f8d416cb0ea3a453e2c9c1ec3b6ef16abf3965d11e437ae82088"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T01:02:31.871783Z","signature_b64":"7drM1DjyIFOlcyyTgnYMjp21OhIb+BJ/fP906zu0rEaMA+Q7yP7k8T/oa6kMVk2sofKEtzvaA9PswNjr8cEdAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"68970c746e5079c1e72e5cfb717a992f10a1141b4c3691b65b8c526c5128ca75","last_reissued_at":"2026-05-18T01:02:31.871173Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T01:02:31.871173Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe, Multi-Agent, Reinforcement Learning for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Amnon Shashua, Shai Shalev-Shwartz, Shaked Shammah","submitted_at":"2016-10-11T12:09:03Z","abstract_excerpt":"Autonomous driving is a multi-agent setting where the host vehicle must apply sophisticated negotiation skills with other road users when overtaking, giving way, merging, taking left and right turns and while pushing ahead in unstructured urban roadways. Since there are many possible scenarios, manually tackling all possible cases will likely yield a too simplistic policy. Moreover, one must balance between unexpected behavior of other drivers/pedestrians and at the same time not to be too defensive so that normal traffic flow is maintained.\n  In this paper we apply deep reinforcement learning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1610.03295","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1610.03295","created_at":"2026-05-18T01:02:31.871260+00:00"},{"alias_kind":"arxiv_version","alias_value":"1610.03295v1","created_at":"2026-05-18T01:02:31.871260+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1610.03295","created_at":"2026-05-18T01:02:31.871260+00:00"},{"alias_kind":"pith_short_12","alias_value":"NCLQY5DOKB44","created_at":"2026-05-18T12:30:32.724797+00:00"},{"alias_kind":"pith_short_16","alias_value":"NCLQY5DOKB44DZZO","created_at":"2026-05-18T12:30:32.724797+00:00"},{"alias_kind":"pith_short_8","alias_value":"NCLQY5DO","created_at":"2026-05-18T12:30:32.724797+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":11,"sample":[{"citing_arxiv_id":"2607.01721","citing_title":"CoRe: Combined Rewards with Vision-Language Model Feedback for Preference-Aligned Reinforcement Learning","ref_index":117,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11266","citing_title":"Seeing Before Colliding: Anticipatory Safe RL with Frozen Vision-Language Models","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2403.10559","citing_title":"Generative Models and Connected and Automated Vehicles: A Survey in Exploring the Intersection of Transportation and AI","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2412.18208","citing_title":"Quantum framework for Reinforcement Learning: Integrating Markov decision process, quantum arithmetic, and trajectory search","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2502.03506","citing_title":"Optimistic {\\epsilon}-Greedy Exploration for Cooperative Multi-Agent Reinforcement Learning","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2506.14648","citing_title":"SENIOR: Efficient Query Selection and Preference-Guided Exploration in Preference-based Reinforcement Learning","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2509.08933","citing_title":"Corruption-Tolerant Asynchronous Q-Learning with Near-Optimal Rates","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20999","citing_title":"Concentration of General Stochastic Approximation Under Heavy-Tailed Markovian Noise","ref_index":81,"is_internal_anchor":true},{"citing_arxiv_id":"2506.19417","citing_title":"Focusing Influence Mechanism for Multi-Agent Reinforcement Learning","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2509.16002","citing_title":"Scalable Quantum Reinforcement Learning on NISQ Devices with Dynamic-Circuit Qubit Reuse and Grover Optimization","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2603.06977","citing_title":"NePPO: Near-Potential Policy Optimization for General-Sum Multi-Agent Reinforcement Learning","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2604.21941","citing_title":"When Altruism Meets Autonomy: Managing Bottleneck Congestion with Strategic Autonomous Vehicles","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12857","citing_title":"Artificial Intelligence for Modeling and Simulation of Mixed Automated and Human Traffic","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4","json":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4.json","graph_json":"https://pith.science/api/pith-number/NCLQY5DOKB44DZZOLT5XC6UZF4/graph.json","events_json":"https://pith.science/api/pith-number/NCLQY5DOKB44DZZOLT5XC6UZF4/events.json","paper":"https://pith.science/paper/NCLQY5DO"},"agent_actions":{"view_html":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4","download_json":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4.json","view_paper":"https://pith.science/paper/NCLQY5DO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1610.03295&json=true","fetch_graph":"https://pith.science/api/pith-number/NCLQY5DOKB44DZZOLT5XC6UZF4/graph.json","fetch_events":"https://pith.science/api/pith-number/NCLQY5DOKB44DZZOLT5XC6UZF4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4/action/storage_attestation","attest_author":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4/action/author_attestation","sign_citation":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4/action/citation_signature","submit_replication":"https://pith.science/pith/NCLQY5DOKB44DZZOLT5XC6UZF4/action/replication_record"}},"created_at":"2026-05-18T01:02:31.871260+00:00","updated_at":"2026-05-18T01:02:31.871260+00:00"}