{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:4LGG7UJLYM4VCLADZ7TXEKOEBM","short_pith_number":"pith:4LGG7UJL","schema_version":"1.0","canonical_sha256":"e2cc6fd12bc339512c03cfe77229c40b3b7f1675393b5ff10254b30c2fb15929","source":{"kind":"arxiv","id":"1906.06639","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning Driven Heuristic Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Azalia Mirhoseini, George Tucker, Jingtao Wang, Qingpeng Cai, Wei Wei, Will Hang","submitted_at":"2019-06-16T03:28:24Z","abstract_excerpt":"Heuristic algorithms such as simulated annealing, Concorde, and METIS are effective and widely used approaches to find solutions to combinatorial optimization problems. However, they are limited by the high sample complexity required to reach a reasonable solution from a cold-start. In this paper, we introduce a novel framework to generate better initial solutions for heuristic algorithms using reinforcement learning (RL), named RLHO. We augment the ability of heuristic algorithms to greedily improve upon an existing initial solution generated by RL, and demonstrate novel results where RL is a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.06639","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-16T03:28:24Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"0469d0231f24c8cdb6c7660ad4e88ecdecb04a0845ca5f00380696e89517d30d","abstract_canon_sha256":"be4f0a7900a57f190230ec25d5319445c1de4cfa45ccb9abfd937869fcb63673"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:43:13.089899Z","signature_b64":"45gnwdbxzZcl2IwPKyZ0UYz8wymZx29m+dqrJ50EDtpjSYGg+yw7k4W1tG/IUc1b3pxqKpZMgS/W6/6I0GQrCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e2cc6fd12bc339512c03cfe77229c40b3b7f1675393b5ff10254b30c2fb15929","last_reissued_at":"2026-05-17T23:43:13.089382Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:43:13.089382Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning Driven Heuristic Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Azalia Mirhoseini, George Tucker, Jingtao Wang, Qingpeng Cai, Wei Wei, Will Hang","submitted_at":"2019-06-16T03:28:24Z","abstract_excerpt":"Heuristic algorithms such as simulated annealing, Concorde, and METIS are effective and widely used approaches to find solutions to combinatorial optimization problems. However, they are limited by the high sample complexity required to reach a reasonable solution from a cold-start. In this paper, we introduce a novel framework to generate better initial solutions for heuristic algorithms using reinforcement learning (RL), named RLHO. We augment the ability of heuristic algorithms to greedily improve upon an existing initial solution generated by RL, and demonstrate novel results where RL is a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.06639","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.06639","created_at":"2026-05-17T23:43:13.089467+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.06639v1","created_at":"2026-05-17T23:43:13.089467+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.06639","created_at":"2026-05-17T23:43:13.089467+00:00"},{"alias_kind":"pith_short_12","alias_value":"4LGG7UJLYM4V","created_at":"2026-05-18T12:33:10.108867+00:00"},{"alias_kind":"pith_short_16","alias_value":"4LGG7UJLYM4VCLAD","created_at":"2026-05-18T12:33:10.108867+00:00"},{"alias_kind":"pith_short_8","alias_value":"4LGG7UJL","created_at":"2026-05-18T12:33:10.108867+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.09404","citing_title":"Synergizing Reinforcement Learning and Genetic Algorithms for Neural Combinatorial Optimization","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM","json":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM.json","graph_json":"https://pith.science/api/pith-number/4LGG7UJLYM4VCLADZ7TXEKOEBM/graph.json","events_json":"https://pith.science/api/pith-number/4LGG7UJLYM4VCLADZ7TXEKOEBM/events.json","paper":"https://pith.science/paper/4LGG7UJL"},"agent_actions":{"view_html":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM","download_json":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM.json","view_paper":"https://pith.science/paper/4LGG7UJL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.06639&json=true","fetch_graph":"https://pith.science/api/pith-number/4LGG7UJLYM4VCLADZ7TXEKOEBM/graph.json","fetch_events":"https://pith.science/api/pith-number/4LGG7UJLYM4VCLADZ7TXEKOEBM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM/action/storage_attestation","attest_author":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM/action/author_attestation","sign_citation":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM/action/citation_signature","submit_replication":"https://pith.science/pith/4LGG7UJLYM4VCLADZ7TXEKOEBM/action/replication_record"}},"created_at":"2026-05-17T23:43:13.089467+00:00","updated_at":"2026-05-17T23:43:13.089467+00:00"}