{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:MFKKA2JMUS7QF4QRBF7EVD6ZN6","short_pith_number":"pith:MFKKA2JM","schema_version":"1.0","canonical_sha256":"6154a0692ca4bf02f211097e4a8fd96f9ea2f7eda94c0114c86982752914caee","source":{"kind":"arxiv","id":"2604.25191","version":2},"attestation_state":"computed","paper":{"title":"How Can Reinforcement Learning Achieve Expert-level Placement?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Reinforcement learning reaches expert chip placement quality by learning a reward model directly from final expert layouts.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.AR","authors_text":"Chao Qian, Chengrui Gao, Ke Xue, Mingxuan Yuan, Peng Xie, Ruo-Tong Chen, Siyuan Xu, Tian Xu, Yunqi Shi, Zhi-Hua Zhou","submitted_at":"2026-04-28T03:55:03Z","abstract_excerpt":"Chip placement is a critical step in physical design. While reinforcement learning (RL)-based methods have recently emerged, their training primarily focuses on wirelength optimization, and therefore often fail to achieve expert-quality layouts. We identify the reward design as the primary cause for the performance gap with experts, and instead of formalizing intricate processes, we circumvent this by directly learning from expert layouts to derive a reward model. Our approach starts from the final expert layouts to infer step-by-step expert trajectories. Using these trajectories as demonstrat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2604.25191","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AR","submitted_at":"2026-04-28T03:55:03Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"5c337aab732a35bda80d51e1390f69cf053d3d93f7a6d2e100aed2bf1a5b3723","abstract_canon_sha256":"86b64e6e00286b16d4046569d3ce6097cf6e6243df04765c74f591b340f1b36d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-02T03:04:41.575984Z","signature_b64":"NySc4YiCAziyk8itR4rubODiwxPuRVVwLKMKFPNqprAjIU+coPbbKZDp7Z0G/eyxjCFikzGD6HlRnTLBWlkWBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6154a0692ca4bf02f211097e4a8fd96f9ea2f7eda94c0114c86982752914caee","last_reissued_at":"2026-06-02T03:04:41.575556Z","signature_status":"signed_v1","first_computed_at":"2026-06-02T03:04:41.575556Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Can Reinforcement Learning Achieve Expert-level Placement?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Reinforcement learning reaches expert chip placement quality by learning a reward model directly from final expert layouts.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.AR","authors_text":"Chao Qian, Chengrui Gao, Ke Xue, Mingxuan Yuan, Peng Xie, Ruo-Tong Chen, Siyuan Xu, Tian Xu, Yunqi Shi, Zhi-Hua Zhou","submitted_at":"2026-04-28T03:55:03Z","abstract_excerpt":"Chip placement is a critical step in physical design. While reinforcement learning (RL)-based methods have recently emerged, their training primarily focuses on wirelength optimization, and therefore often fail to achieve expert-quality layouts. We identify the reward design as the primary cause for the performance gap with experts, and instead of formalizing intricate processes, we circumvent this by directly learning from expert layouts to derive a reward model. Our approach starts from the final expert layouts to infer step-by-step expert trajectories. Using these trajectories as demonstrat"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Experiments show that our framework can efficiently learn from even a single design and generalize well to unseen cases.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That step-by-step expert trajectories can be reliably inferred from final layouts alone without additional information about the expert's intermediate decisions or constraints.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"RL chip placement learns an implicit reward model from expert trajectories inferred from final layouts, closing the gap to human experts even from a single design.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Reinforcement learning reaches expert chip placement quality by learning a reward model directly from final expert layouts.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"a435f0a75b1c946e3a45894ecb5d2db83b4985d85677c4bb0075870cf9220b8b"},"source":{"id":"2604.25191","kind":"arxiv","version":2},"verdict":{"id":"99f7ee32-f429-485e-9b8e-3dbae8bc49e6","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-07T14:38:20.956719Z","strongest_claim":"Experiments show that our framework can efficiently learn from even a single design and generalize well to unseen cases.","one_line_summary":"RL chip placement learns an implicit reward model from expert trajectories inferred from final layouts, closing the gap to human experts even from a single design.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That step-by-step expert trajectories can be reliably inferred from final layouts alone without additional information about the expert's intermediate decisions or constraints.","pith_extraction_headline":"Reinforcement learning reaches expert chip placement quality by learning a reward model directly from final expert layouts."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.25191/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-21T05:37:48.179156Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T21:23:52.068686Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"489109203befe3de7f0c7b1e9873ac5adfb96e387eec62b55244ecc8c819d041"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2604.25191","created_at":"2026-06-02T03:04:41.575608+00:00"},{"alias_kind":"arxiv_version","alias_value":"2604.25191v2","created_at":"2026-06-02T03:04:41.575608+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.25191","created_at":"2026-06-02T03:04:41.575608+00:00"},{"alias_kind":"pith_short_12","alias_value":"MFKKA2JMUS7Q","created_at":"2026-06-02T03:04:41.575608+00:00"},{"alias_kind":"pith_short_16","alias_value":"MFKKA2JMUS7QF4QR","created_at":"2026-06-02T03:04:41.575608+00:00"},{"alias_kind":"pith_short_8","alias_value":"MFKKA2JM","created_at":"2026-06-02T03:04:41.575608+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6","json":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6.json","graph_json":"https://pith.science/api/pith-number/MFKKA2JMUS7QF4QRBF7EVD6ZN6/graph.json","events_json":"https://pith.science/api/pith-number/MFKKA2JMUS7QF4QRBF7EVD6ZN6/events.json","paper":"https://pith.science/paper/MFKKA2JM"},"agent_actions":{"view_html":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6","download_json":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6.json","view_paper":"https://pith.science/paper/MFKKA2JM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2604.25191&json=true","fetch_graph":"https://pith.science/api/pith-number/MFKKA2JMUS7QF4QRBF7EVD6ZN6/graph.json","fetch_events":"https://pith.science/api/pith-number/MFKKA2JMUS7QF4QRBF7EVD6ZN6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6/action/storage_attestation","attest_author":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6/action/author_attestation","sign_citation":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6/action/citation_signature","submit_replication":"https://pith.science/pith/MFKKA2JMUS7QF4QRBF7EVD6ZN6/action/replication_record"}},"created_at":"2026-06-02T03:04:41.575608+00:00","updated_at":"2026-06-02T03:04:41.575608+00:00"}