{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KDQNKFPEKG5T36MCYSPLF4KONQ","short_pith_number":"pith:KDQNKFPE","schema_version":"1.0","canonical_sha256":"50e0d515e451bb3df982c49eb2f14e6c3590916f1e22dd290edda97acc879f6f","source":{"kind":"arxiv","id":"2506.10764","version":1},"attestation_state":"computed","paper":{"title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Haodong Duan, Jixuan Chen, Kai Chen, Qingwen Liu, Shengyuan Ding, Xiaozhe Li, Xinyu Fang","submitted_at":"2025-06-12T14:46:41Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in solving diverse tasks. However, their proficiency in iteratively optimizing complex solutions through learning from previous feedback remains insufficiently explored. To bridge this gap, we present OPT-BENCH, a comprehensive benchmark designed to evaluate LLM agents on large-scale search space optimization problems. OPT-BENCH includes 20 real-world machine learning tasks sourced from Kaggle and 10 classical NP problems, offering a diverse and challenging environment for assessing LLM agents on iterative reasoning and solution r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.10764","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-12T14:46:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"03e46f99fcc901dcf4411b8b4163626fe40e41c48cc15b3a1260c224d8cff6f9","abstract_canon_sha256":"945d9f07ef20baa8c10c288997617bc2b862c1e574b0e4e3ca302122133fe6c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:30.181246Z","signature_b64":"m15hu4R8y4d68904rNM41F9lZRQUJdGkCugvD9u9VMQZy9SDBZbdvL1/8sEf2faub1l8Xo4YiBZ/LqSHAcLyBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50e0d515e451bb3df982c49eb2f14e6c3590916f1e22dd290edda97acc879f6f","last_reissued_at":"2026-07-05T11:20:30.180732Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:30.180732Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Haodong Duan, Jixuan Chen, Kai Chen, Qingwen Liu, Shengyuan Ding, Xiaozhe Li, Xinyu Fang","submitted_at":"2025-06-12T14:46:41Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in solving diverse tasks. However, their proficiency in iteratively optimizing complex solutions through learning from previous feedback remains insufficiently explored. To bridge this gap, we present OPT-BENCH, a comprehensive benchmark designed to evaluate LLM agents on large-scale search space optimization problems. OPT-BENCH includes 20 real-world machine learning tasks sourced from Kaggle and 10 classical NP problems, offering a diverse and challenging environment for assessing LLM agents on iterative reasoning and solution r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.10764","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.10764/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.10764","created_at":"2026-07-05T11:20:30.180795+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.10764v1","created_at":"2026-07-05T11:20:30.180795+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.10764","created_at":"2026-07-05T11:20:30.180795+00:00"},{"alias_kind":"pith_short_12","alias_value":"KDQNKFPEKG5T","created_at":"2026-07-05T11:20:30.180795+00:00"},{"alias_kind":"pith_short_16","alias_value":"KDQNKFPEKG5T36MC","created_at":"2026-07-05T11:20:30.180795+00:00"},{"alias_kind":"pith_short_8","alias_value":"KDQNKFPE","created_at":"2026-07-05T11:20:30.180795+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25832","citing_title":"MiniOpt: Reasoning to Model and Solve General Optimization Problems with Limited Resources","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25832","citing_title":"MiniOpt: Reasoning to Model and Solve General Optimization Problems with Limited Resources","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19338","citing_title":"Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20849","citing_title":"Large Language Models for Operations Research: A Comprehensive Survey","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19447","citing_title":"What and When to Distill: Selective Hindsight Distillation for Multi-Turn Agents","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08905","citing_title":"Forge: Quality-Aware Reinforcement Learning for NP-Hard Optimization in LLMs","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19440","citing_title":"What Makes an LLM a Good Optimizer? A Trajectory Analysis of LLM-Guided Evolutionary Search","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ","json":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ.json","graph_json":"https://pith.science/api/pith-number/KDQNKFPEKG5T36MCYSPLF4KONQ/graph.json","events_json":"https://pith.science/api/pith-number/KDQNKFPEKG5T36MCYSPLF4KONQ/events.json","paper":"https://pith.science/paper/KDQNKFPE"},"agent_actions":{"view_html":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ","download_json":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ.json","view_paper":"https://pith.science/paper/KDQNKFPE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.10764&json=true","fetch_graph":"https://pith.science/api/pith-number/KDQNKFPEKG5T36MCYSPLF4KONQ/graph.json","fetch_events":"https://pith.science/api/pith-number/KDQNKFPEKG5T36MCYSPLF4KONQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ/action/storage_attestation","attest_author":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ/action/author_attestation","sign_citation":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ/action/citation_signature","submit_replication":"https://pith.science/pith/KDQNKFPEKG5T36MCYSPLF4KONQ/action/replication_record"}},"created_at":"2026-07-05T11:20:30.180795+00:00","updated_at":"2026-07-05T11:20:30.180795+00:00"}