{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TFWSI54YVJJNX3UBCT3OSPDJUM","short_pith_number":"pith:TFWSI54Y","schema_version":"1.0","canonical_sha256":"996d247798aa52dbee8114f6e93c69a32a92a23658ee5c43a6153674a9093938","source":{"kind":"arxiv","id":"2409.19924","version":4},"attestation_state":"computed","paper":{"title":"On The Planning Abilities of OpenAI's o1 Models: Feasibility, Optimality, and Generalizability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.AI","authors_text":"Junbo Li, Kevin Wang, Neel P. Bhatt, Qiang Liu, Ufuk Topcu, Yihan Xi, Zhangyang Wang","submitted_at":"2024-09-30T03:58:43Z","abstract_excerpt":"Recent advancements in Large Language Models (LLMs) have showcased their ability to perform complex reasoning tasks, but their effectiveness in planning remains underexplored. In this study, we evaluate the planning capabilities of OpenAI's o1 models across a variety of benchmark tasks, focusing on three key aspects: feasibility, optimality, and generalizability. Through empirical evaluations on constraint-heavy tasks (e.g., $\\textit{Barman}$, $\\textit{Tyreworld}$) and spatially complex environments (e.g., $\\textit{Termes}$, $\\textit{Floortile}$), we highlight o1-preview's strengths in self-ev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.19924","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-09-30T03:58:43Z","cross_cats_sorted":["cs.LG","cs.RO"],"title_canon_sha256":"7564bed15145b1acee3e94a1b3fbae2284056d2120e6ae280819a62f6428556c","abstract_canon_sha256":"4c2be841393f1bf9293f71933286c5be1e011a132088de8d68d14352e5fd52f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:03.611153Z","signature_b64":"I7mNk4JLFwyex8f73LdREnJSz4wrBbk7iLioV0Wb9oO+Q9ZMAKUFZPTid6h2G2JzxScit0usApii2f+M6o/EDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"996d247798aa52dbee8114f6e93c69a32a92a23658ee5c43a6153674a9093938","last_reissued_at":"2026-07-05T09:20:03.610739Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:03.610739Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On The Planning Abilities of OpenAI's o1 Models: Feasibility, Optimality, and Generalizability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.AI","authors_text":"Junbo Li, Kevin Wang, Neel P. Bhatt, Qiang Liu, Ufuk Topcu, Yihan Xi, Zhangyang Wang","submitted_at":"2024-09-30T03:58:43Z","abstract_excerpt":"Recent advancements in Large Language Models (LLMs) have showcased their ability to perform complex reasoning tasks, but their effectiveness in planning remains underexplored. In this study, we evaluate the planning capabilities of OpenAI's o1 models across a variety of benchmark tasks, focusing on three key aspects: feasibility, optimality, and generalizability. Through empirical evaluations on constraint-heavy tasks (e.g., $\\textit{Barman}$, $\\textit{Tyreworld}$) and spatially complex environments (e.g., $\\textit{Termes}$, $\\textit{Floortile}$), we highlight o1-preview's strengths in self-ev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.19924","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.19924/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.19924","created_at":"2026-07-05T09:20:03.610800+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.19924v4","created_at":"2026-07-05T09:20:03.610800+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.19924","created_at":"2026-07-05T09:20:03.610800+00:00"},{"alias_kind":"pith_short_12","alias_value":"TFWSI54YVJJN","created_at":"2026-07-05T09:20:03.610800+00:00"},{"alias_kind":"pith_short_16","alias_value":"TFWSI54YVJJNX3UB","created_at":"2026-07-05T09:20:03.610800+00:00"},{"alias_kind":"pith_short_8","alias_value":"TFWSI54Y","created_at":"2026-07-05T09:20:03.610800+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.09686","citing_title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","ref_index":148,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM","json":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM.json","graph_json":"https://pith.science/api/pith-number/TFWSI54YVJJNX3UBCT3OSPDJUM/graph.json","events_json":"https://pith.science/api/pith-number/TFWSI54YVJJNX3UBCT3OSPDJUM/events.json","paper":"https://pith.science/paper/TFWSI54Y"},"agent_actions":{"view_html":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM","download_json":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM.json","view_paper":"https://pith.science/paper/TFWSI54Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.19924&json=true","fetch_graph":"https://pith.science/api/pith-number/TFWSI54YVJJNX3UBCT3OSPDJUM/graph.json","fetch_events":"https://pith.science/api/pith-number/TFWSI54YVJJNX3UBCT3OSPDJUM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM/action/storage_attestation","attest_author":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM/action/author_attestation","sign_citation":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM/action/citation_signature","submit_replication":"https://pith.science/pith/TFWSI54YVJJNX3UBCT3OSPDJUM/action/replication_record"}},"created_at":"2026-07-05T09:20:03.610800+00:00","updated_at":"2026-07-05T09:20:03.610800+00:00"}