{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PQJ73JLMEZZ42DNH5GA2HGTN53","short_pith_number":"pith:PQJ73JLM","schema_version":"1.0","canonical_sha256":"7c13fda56c2673cd0da7e981a39a6deef4bbfbc4017cf5d3d057a06ca10ce45c","source":{"kind":"arxiv","id":"2507.00711","version":1},"attestation_state":"computed","paper":{"title":"Large Reasoning Models are not thinking straight: on the unreliability of thinking trajectories","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jhouben Cuesta-Ramirez, Mehdi Mounsif, Samuel Beaussant","submitted_at":"2025-07-01T12:14:22Z","abstract_excerpt":"Large Language Models (LLMs) trained via Reinforcement Learning (RL) have recently achieved impressive results on reasoning benchmarks. Yet, growing evidence shows that these models often generate longer but ineffective chains of thought (CoTs), calling into question whether benchmark gains reflect real reasoning improvements. We present new evidence of overthinking, where models disregard correct solutions even when explicitly provided, instead continuing to generate unnecessary reasoning steps that often lead to incorrect conclusions. Experiments on three state-of-the-art models using the AI"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.00711","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-01T12:14:22Z","cross_cats_sorted":[],"title_canon_sha256":"6dc84c48de4a93e22f685e07ba9e90922ba99736707c9186c9af7ffaa33e8b2b","abstract_canon_sha256":"a5bad1911fd1c0c6435df51df0b7eef1b141371bd29b0a525b901d698139ac5e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:13.353916Z","signature_b64":"eKj3UIVTSlU4yxKEL8FG1RW3MfWznUOg0n7bsgarb1BDz1RdnlWz5dTrI+n2VoN5aAZN5lPfMP1hLMv8PA6zDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c13fda56c2673cd0da7e981a39a6deef4bbfbc4017cf5d3d057a06ca10ce45c","last_reissued_at":"2026-07-05T11:30:13.353448Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:13.353448Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Reasoning Models are not thinking straight: on the unreliability of thinking trajectories","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jhouben Cuesta-Ramirez, Mehdi Mounsif, Samuel Beaussant","submitted_at":"2025-07-01T12:14:22Z","abstract_excerpt":"Large Language Models (LLMs) trained via Reinforcement Learning (RL) have recently achieved impressive results on reasoning benchmarks. Yet, growing evidence shows that these models often generate longer but ineffective chains of thought (CoTs), calling into question whether benchmark gains reflect real reasoning improvements. We present new evidence of overthinking, where models disregard correct solutions even when explicitly provided, instead continuing to generate unnecessary reasoning steps that often lead to incorrect conclusions. Experiments on three state-of-the-art models using the AI"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00711","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.00711/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.00711","created_at":"2026-07-05T11:30:13.353507+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.00711v1","created_at":"2026-07-05T11:30:13.353507+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00711","created_at":"2026-07-05T11:30:13.353507+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQJ73JLMEZZ4","created_at":"2026-07-05T11:30:13.353507+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQJ73JLMEZZ42DNH","created_at":"2026-07-05T11:30:13.353507+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQJ73JLM","created_at":"2026-07-05T11:30:13.353507+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22211","citing_title":"CLORE: Content-Level Optimization for Reasoning Efficiency","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53","json":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53.json","graph_json":"https://pith.science/api/pith-number/PQJ73JLMEZZ42DNH5GA2HGTN53/graph.json","events_json":"https://pith.science/api/pith-number/PQJ73JLMEZZ42DNH5GA2HGTN53/events.json","paper":"https://pith.science/paper/PQJ73JLM"},"agent_actions":{"view_html":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53","download_json":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53.json","view_paper":"https://pith.science/paper/PQJ73JLM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.00711&json=true","fetch_graph":"https://pith.science/api/pith-number/PQJ73JLMEZZ42DNH5GA2HGTN53/graph.json","fetch_events":"https://pith.science/api/pith-number/PQJ73JLMEZZ42DNH5GA2HGTN53/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53/action/storage_attestation","attest_author":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53/action/author_attestation","sign_citation":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53/action/citation_signature","submit_replication":"https://pith.science/pith/PQJ73JLMEZZ42DNH5GA2HGTN53/action/replication_record"}},"created_at":"2026-07-05T11:30:13.353507+00:00","updated_at":"2026-07-05T11:30:13.353507+00:00"}