{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FXKPT6A5BLEPOAQWB7YOVNPAJA","short_pith_number":"pith:FXKPT6A5","schema_version":"1.0","canonical_sha256":"2dd4f9f81d0ac8f702160ff0eab5e04834e2e52002cf938bed6e97f0eec6942f","source":{"kind":"arxiv","id":"2404.11041","version":2},"attestation_state":"computed","paper":{"title":"On the Empirical Complexity of Reasoning and Planning in LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"David Hsu, Liwei Kang, Wee Sun Lee, Zirui Zhao","submitted_at":"2024-04-17T03:34:27Z","abstract_excerpt":"Chain-of-thought (CoT), tree-of-thought (ToT), and related techniques work surprisingly well in practice for some complex reasoning tasks with Large Language Models (LLMs), but why? This work seeks the underlying reasons by conducting experimental case studies and linking the performance benefits to well-established sample and computational complexity principles in machine learning. We experimented with 6 reasoning tasks, ranging from grade school math, air travel planning, ..., to Blocksworld. The results suggest that (i) both CoT and ToT benefit significantly from task decomposition, which b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.11041","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-04-17T03:34:27Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"060495add59d9b462095c4e72e9f91c966024046f805dbac9df60c3b6701faaf","abstract_canon_sha256":"2d0a62eeb5cd8266a14720eb65188fdd3d58442db7ceb5bbe3026c484e5edac3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:39.056552Z","signature_b64":"rS6+BTVL1a6eTNOC5PkeNEXcjgGaVnFiYVdAfqooIxTGzo7VbYKgt34UKEewHZ3afSj4OkSUlNc6Qs38VAC2AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dd4f9f81d0ac8f702160ff0eab5e04834e2e52002cf938bed6e97f0eec6942f","last_reissued_at":"2026-07-05T08:33:39.055957Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:39.055957Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Empirical Complexity of Reasoning and Planning in LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"David Hsu, Liwei Kang, Wee Sun Lee, Zirui Zhao","submitted_at":"2024-04-17T03:34:27Z","abstract_excerpt":"Chain-of-thought (CoT), tree-of-thought (ToT), and related techniques work surprisingly well in practice for some complex reasoning tasks with Large Language Models (LLMs), but why? This work seeks the underlying reasons by conducting experimental case studies and linking the performance benefits to well-established sample and computational complexity principles in machine learning. We experimented with 6 reasoning tasks, ranging from grade school math, air travel planning, ..., to Blocksworld. The results suggest that (i) both CoT and ToT benefit significantly from task decomposition, which b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.11041","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.11041/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.11041","created_at":"2026-07-05T08:33:39.056026+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.11041v2","created_at":"2026-07-05T08:33:39.056026+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.11041","created_at":"2026-07-05T08:33:39.056026+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXKPT6A5BLEP","created_at":"2026-07-05T08:33:39.056026+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXKPT6A5BLEPOAQW","created_at":"2026-07-05T08:33:39.056026+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXKPT6A5","created_at":"2026-07-05T08:33:39.056026+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.04098","citing_title":"TextAtari: 100K Frames Game Playing with Language Agents","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA","json":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA.json","graph_json":"https://pith.science/api/pith-number/FXKPT6A5BLEPOAQWB7YOVNPAJA/graph.json","events_json":"https://pith.science/api/pith-number/FXKPT6A5BLEPOAQWB7YOVNPAJA/events.json","paper":"https://pith.science/paper/FXKPT6A5"},"agent_actions":{"view_html":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA","download_json":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA.json","view_paper":"https://pith.science/paper/FXKPT6A5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.11041&json=true","fetch_graph":"https://pith.science/api/pith-number/FXKPT6A5BLEPOAQWB7YOVNPAJA/graph.json","fetch_events":"https://pith.science/api/pith-number/FXKPT6A5BLEPOAQWB7YOVNPAJA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA/action/storage_attestation","attest_author":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA/action/author_attestation","sign_citation":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA/action/citation_signature","submit_replication":"https://pith.science/pith/FXKPT6A5BLEPOAQWB7YOVNPAJA/action/replication_record"}},"created_at":"2026-07-05T08:33:39.056026+00:00","updated_at":"2026-07-05T08:33:39.056026+00:00"}