{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YX4QNJBRI72YTW72TUQRK2DHMV","short_pith_number":"pith:YX4QNJBR","schema_version":"1.0","canonical_sha256":"c5f906a43147f589dbfa9d21156867657cbc73bae24d2556981dff58d4aebe98","source":{"kind":"arxiv","id":"2504.19480","version":1},"attestation_state":"computed","paper":{"title":"An Automated Reinforcement Learning Reward Design Framework with Large Language Model for Cooperative Platoon Coordination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dixiao Wei, Jinlong Lei, Peng Yi, Yiguang Hong, Yuchuan Du","submitted_at":"2025-04-28T04:41:15Z","abstract_excerpt":"Reinforcement Learning (RL) has demonstrated excellent decision-making potential in platoon coordination problems. However, due to the variability of coordination goals, the complexity of the decision problem, and the time-consumption of trial-and-error in manual design, finding a well performance reward function to guide RL training to solve complex platoon coordination problems remains challenging. In this paper, we formally define the Platoon Coordination Reward Design Problem (PCRDP), extending the RL-based cooperative platoon coordination problem to incorporate automated reward function g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.19480","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-28T04:41:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"42ab6d0cbc092fbfb366f356900c5682d89c04cbde761754b90ec0060cd03733","abstract_canon_sha256":"87740598f6c91f64b1e7efbee63ec09ddc2b70bde46839321b4215dd63c56e73"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:58.208302Z","signature_b64":"tMjMJebqxQcurv1XxLH53WnjoRjhPHEK0lpgtM4G5iSVB76jutfpL1u/k2zhubz8ftVoWXqLnlyuvtnfweUSDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5f906a43147f589dbfa9d21156867657cbc73bae24d2556981dff58d4aebe98","last_reissued_at":"2026-07-05T10:54:58.207839Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:58.207839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Automated Reinforcement Learning Reward Design Framework with Large Language Model for Cooperative Platoon Coordination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dixiao Wei, Jinlong Lei, Peng Yi, Yiguang Hong, Yuchuan Du","submitted_at":"2025-04-28T04:41:15Z","abstract_excerpt":"Reinforcement Learning (RL) has demonstrated excellent decision-making potential in platoon coordination problems. However, due to the variability of coordination goals, the complexity of the decision problem, and the time-consumption of trial-and-error in manual design, finding a well performance reward function to guide RL training to solve complex platoon coordination problems remains challenging. In this paper, we formally define the Platoon Coordination Reward Design Problem (PCRDP), extending the RL-based cooperative platoon coordination problem to incorporate automated reward function g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.19480","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.19480/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.19480","created_at":"2026-07-05T10:54:58.207895+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.19480v1","created_at":"2026-07-05T10:54:58.207895+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.19480","created_at":"2026-07-05T10:54:58.207895+00:00"},{"alias_kind":"pith_short_12","alias_value":"YX4QNJBRI72Y","created_at":"2026-07-05T10:54:58.207895+00:00"},{"alias_kind":"pith_short_16","alias_value":"YX4QNJBRI72YTW72","created_at":"2026-07-05T10:54:58.207895+00:00"},{"alias_kind":"pith_short_8","alias_value":"YX4QNJBR","created_at":"2026-07-05T10:54:58.207895+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV","json":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV.json","graph_json":"https://pith.science/api/pith-number/YX4QNJBRI72YTW72TUQRK2DHMV/graph.json","events_json":"https://pith.science/api/pith-number/YX4QNJBRI72YTW72TUQRK2DHMV/events.json","paper":"https://pith.science/paper/YX4QNJBR"},"agent_actions":{"view_html":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV","download_json":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV.json","view_paper":"https://pith.science/paper/YX4QNJBR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.19480&json=true","fetch_graph":"https://pith.science/api/pith-number/YX4QNJBRI72YTW72TUQRK2DHMV/graph.json","fetch_events":"https://pith.science/api/pith-number/YX4QNJBRI72YTW72TUQRK2DHMV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV/action/storage_attestation","attest_author":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV/action/author_attestation","sign_citation":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV/action/citation_signature","submit_replication":"https://pith.science/pith/YX4QNJBRI72YTW72TUQRK2DHMV/action/replication_record"}},"created_at":"2026-07-05T10:54:58.207895+00:00","updated_at":"2026-07-05T10:54:58.207895+00:00"}