{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UTEDE6PPJKBAFFTP5O5I2GI5Q5","short_pith_number":"pith:UTEDE6PP","schema_version":"1.0","canonical_sha256":"a4c83279ef4a8202966febba8d191d8755fb38ad4cd475ed45cc32f9824d68c8","source":{"kind":"arxiv","id":"2506.09250","version":2},"attestation_state":"computed","paper":{"title":"Comment on The Illusion of Thinking: Understanding the Strengths and Limitations of Reasoning Models via the Lens of Problem Complexity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"A. Lawsen","submitted_at":"2025-06-10T21:16:53Z","abstract_excerpt":"Shojaee et al. (2025) report that Large Reasoning Models (LRMs) exhibit \"accuracy collapse\" on planning puzzles beyond certain complexity thresholds. We demonstrate that their findings primarily reflect experimental design limitations rather than fundamental reasoning failures. Our analysis reveals three critical issues: (1) Tower of Hanoi experiments risk exceeding model output token limits, with models explicitly acknowledging these constraints in their outputs; (2) The authors' automated evaluation framework fails to distinguish between reasoning failures and practical constraints, leading "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09250","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T21:16:53Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"da24e437e967bc44e557da492c13e1a54dd8fb1eac990529e0d974278c68ca25","abstract_canon_sha256":"37621a46aef3abd8429db8be2782d841a4cb87c87e8779160fa2b1e931a5887f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:41.731973Z","signature_b64":"ELs7Z9VYjgfDyaO+sbxqR7sZ9dpWPk0A9nOxEaudEyyld2xq5LAHHjLciKUmZjeZjQK5lzOON5A9uxk9AFbRCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a4c83279ef4a8202966febba8d191d8755fb38ad4cd475ed45cc32f9824d68c8","last_reissued_at":"2026-07-05T11:22:41.731354Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:41.731354Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Comment on The Illusion of Thinking: Understanding the Strengths and Limitations of Reasoning Models via the Lens of Problem Complexity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"A. Lawsen","submitted_at":"2025-06-10T21:16:53Z","abstract_excerpt":"Shojaee et al. (2025) report that Large Reasoning Models (LRMs) exhibit \"accuracy collapse\" on planning puzzles beyond certain complexity thresholds. We demonstrate that their findings primarily reflect experimental design limitations rather than fundamental reasoning failures. Our analysis reveals three critical issues: (1) Tower of Hanoi experiments risk exceeding model output token limits, with models explicitly acknowledging these constraints in their outputs; (2) The authors' automated evaluation framework fails to distinguish between reasoning failures and practical constraints, leading "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09250","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09250/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09250","created_at":"2026-07-05T11:22:41.731404+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09250v2","created_at":"2026-07-05T11:22:41.731404+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09250","created_at":"2026-07-05T11:22:41.731404+00:00"},{"alias_kind":"pith_short_12","alias_value":"UTEDE6PPJKBA","created_at":"2026-07-05T11:22:41.731404+00:00"},{"alias_kind":"pith_short_16","alias_value":"UTEDE6PPJKBAFFTP","created_at":"2026-07-05T11:22:41.731404+00:00"},{"alias_kind":"pith_short_8","alias_value":"UTEDE6PP","created_at":"2026-07-05T11:22:41.731404+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.23108","citing_title":"Artificial Phantasia: Emergent Mental Imagery in Large Language Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04023","citing_title":"Do LLMs Overthink Basic Math Reasoning? Benchmarking the Accuracy-Efficiency Tradeoff in Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21882","citing_title":"Position: The Hidden Costs and Measurement Gaps of Reinforcement Learning with Verifiable Rewards","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06870","citing_title":"LEAD: Breaking the No-Recovery Bottleneck in Long-Horizon Reasoning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5","json":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5.json","graph_json":"https://pith.science/api/pith-number/UTEDE6PPJKBAFFTP5O5I2GI5Q5/graph.json","events_json":"https://pith.science/api/pith-number/UTEDE6PPJKBAFFTP5O5I2GI5Q5/events.json","paper":"https://pith.science/paper/UTEDE6PP"},"agent_actions":{"view_html":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5","download_json":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5.json","view_paper":"https://pith.science/paper/UTEDE6PP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09250&json=true","fetch_graph":"https://pith.science/api/pith-number/UTEDE6PPJKBAFFTP5O5I2GI5Q5/graph.json","fetch_events":"https://pith.science/api/pith-number/UTEDE6PPJKBAFFTP5O5I2GI5Q5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5/action/storage_attestation","attest_author":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5/action/author_attestation","sign_citation":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5/action/citation_signature","submit_replication":"https://pith.science/pith/UTEDE6PPJKBAFFTP5O5I2GI5Q5/action/replication_record"}},"created_at":"2026-07-05T11:22:41.731404+00:00","updated_at":"2026-07-05T11:22:41.731404+00:00"}