{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FS2AJ3EQOC3GPPZAFQ7BW3VDE2","short_pith_number":"pith:FS2AJ3EQ","schema_version":"1.0","canonical_sha256":"2cb404ec9070b667bf202c3e1b6ea326b0485ea5b24ae611b3f2ccabf8e58723","source":{"kind":"arxiv","id":"2501.05057","version":1},"attestation_state":"computed","paper":{"title":"LearningFlow: Automated Policy Learning Workflow for Urban Driving with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Jun Ma, Lei Zheng, Xu Han, Yubin Wang, Zengqi Peng","submitted_at":"2025-01-09T08:28:16Z","abstract_excerpt":"Recent advancements in reinforcement learning (RL) demonstrate the significant potential in autonomous driving. Despite this promise, challenges such as the manual design of reward functions and low sample efficiency in complex environments continue to impede the development of safe and effective driving policies. To tackle these issues, we introduce LearningFlow, an innovative automated policy learning workflow tailored to urban driving. This framework leverages the collaboration of multiple large language model (LLM) agents throughout the RL training process. LearningFlow includes a curricul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05057","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-01-09T08:28:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f373372276e0374e8aac46aab9d6053416eed1c7c7ab3b89fc6dea9a60b27413","abstract_canon_sha256":"8e7105da5cb162caf81e5f8674926b991c8e2a6fd6a6de1d94dc28eb9b6b0998"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:03.651621Z","signature_b64":"LqCDfJwE2x45SERyy2px47nW5a7Zm+lsNofYeoFRvmqqqdPcxPqoU40BJCa7d4kcW20wXtAovVz0WhxZj9PxAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cb404ec9070b667bf202c3e1b6ea326b0485ea5b24ae611b3f2ccabf8e58723","last_reissued_at":"2026-07-05T09:59:03.651177Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:03.651177Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LearningFlow: Automated Policy Learning Workflow for Urban Driving with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Jun Ma, Lei Zheng, Xu Han, Yubin Wang, Zengqi Peng","submitted_at":"2025-01-09T08:28:16Z","abstract_excerpt":"Recent advancements in reinforcement learning (RL) demonstrate the significant potential in autonomous driving. Despite this promise, challenges such as the manual design of reward functions and low sample efficiency in complex environments continue to impede the development of safe and effective driving policies. To tackle these issues, we introduce LearningFlow, an innovative automated policy learning workflow tailored to urban driving. This framework leverages the collaboration of multiple large language model (LLM) agents throughout the RL training process. LearningFlow includes a curricul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05057","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05057/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05057","created_at":"2026-07-05T09:59:03.651236+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05057v1","created_at":"2026-07-05T09:59:03.651236+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05057","created_at":"2026-07-05T09:59:03.651236+00:00"},{"alias_kind":"pith_short_12","alias_value":"FS2AJ3EQOC3G","created_at":"2026-07-05T09:59:03.651236+00:00"},{"alias_kind":"pith_short_16","alias_value":"FS2AJ3EQOC3GPPZA","created_at":"2026-07-05T09:59:03.651236+00:00"},{"alias_kind":"pith_short_8","alias_value":"FS2AJ3EQ","created_at":"2026-07-05T09:59:03.651236+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.15793","citing_title":"HCRMP: A LLM-Hinted Contextual Reinforcement Learning Framework for Autonomous Driving","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2","json":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2.json","graph_json":"https://pith.science/api/pith-number/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/graph.json","events_json":"https://pith.science/api/pith-number/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/events.json","paper":"https://pith.science/paper/FS2AJ3EQ"},"agent_actions":{"view_html":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2","download_json":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2.json","view_paper":"https://pith.science/paper/FS2AJ3EQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05057&json=true","fetch_graph":"https://pith.science/api/pith-number/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/graph.json","fetch_events":"https://pith.science/api/pith-number/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/action/storage_attestation","attest_author":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/action/author_attestation","sign_citation":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/action/citation_signature","submit_replication":"https://pith.science/pith/FS2AJ3EQOC3GPPZAFQ7BW3VDE2/action/replication_record"}},"created_at":"2026-07-05T09:59:03.651236+00:00","updated_at":"2026-07-05T09:59:03.651236+00:00"}