{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:6XH2FXNA7LA3WU2RHBSRT5CRW7","short_pith_number":"pith:6XH2FXNA","schema_version":"1.0","canonical_sha256":"f5cfa2dda0fac1bb5351386519f451b7ff5da89d6dfb385266c4be9ce3b981a3","source":{"kind":"arxiv","id":"2607.22925","version":1},"attestation_state":"computed","paper":{"title":"Not All LLM Reasoning is Visible in the Chain-of-Thought","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ashwinee Panda, Tom Goldstein, Vatsal Baherwani","submitted_at":"2026-07-24T21:32:48Z","abstract_excerpt":"A key question for AI safety is whether a language model expresses all of its reasoning in its output tokens. We demonstrate a concrete failure mode where frontier models exhibit invisible reasoning by leveraging semantically irrelevant filler tokens to improve performance on synthetic reasoning tasks. We evaluate 13 frontier language models across three tasks and find that many models benefit significantly from filler tokens, with accuracy improvements of up to 13 percentage points. The benefit depends on which tokens are used and differs across models. We further show that filler tokens enab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.22925","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-24T21:32:48Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"4143c3272327dd18c2f10b96cc7d0d3d7c0af710e47b62541b02ec8d8badd123","abstract_canon_sha256":"f387e77ab4474cb150e758415b0f5f6dd02f46c473035e008c259eb44f769e9e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T00:22:02.404283Z","signature_b64":"7+rn99woeD/Lny0S2NGPca2YX6THfg2Wninz5rvLBCVajzEC+Lubt1uxZmaxa4Bf9yrdYPrQ+e6/SM/JZzl1DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5cfa2dda0fac1bb5351386519f451b7ff5da89d6dfb385266c4be9ce3b981a3","last_reissued_at":"2026-07-28T00:22:02.403400Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T00:22:02.403400Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not All LLM Reasoning is Visible in the Chain-of-Thought","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ashwinee Panda, Tom Goldstein, Vatsal Baherwani","submitted_at":"2026-07-24T21:32:48Z","abstract_excerpt":"A key question for AI safety is whether a language model expresses all of its reasoning in its output tokens. We demonstrate a concrete failure mode where frontier models exhibit invisible reasoning by leveraging semantically irrelevant filler tokens to improve performance on synthetic reasoning tasks. We evaluate 13 frontier language models across three tasks and find that many models benefit significantly from filler tokens, with accuracy improvements of up to 13 percentage points. The benefit depends on which tokens are used and differs across models. We further show that filler tokens enab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.22925","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.22925/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.22925","created_at":"2026-07-28T00:22:02.403843+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.22925v1","created_at":"2026-07-28T00:22:02.403843+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.22925","created_at":"2026-07-28T00:22:02.403843+00:00"},{"alias_kind":"pith_short_12","alias_value":"6XH2FXNA7LA3","created_at":"2026-07-28T00:22:02.403843+00:00"},{"alias_kind":"pith_short_16","alias_value":"6XH2FXNA7LA3WU2R","created_at":"2026-07-28T00:22:02.403843+00:00"},{"alias_kind":"pith_short_8","alias_value":"6XH2FXNA","created_at":"2026-07-28T00:22:02.403843+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01347","citing_title":"Prompt-Induced Waste in Coding Agents: Reasoning Structure, Tool Behavior, and End-to-End Cost","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7","json":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7.json","graph_json":"https://pith.science/api/pith-number/6XH2FXNA7LA3WU2RHBSRT5CRW7/graph.json","events_json":"https://pith.science/api/pith-number/6XH2FXNA7LA3WU2RHBSRT5CRW7/events.json","paper":"https://pith.science/paper/6XH2FXNA"},"agent_actions":{"view_html":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7","download_json":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7.json","view_paper":"https://pith.science/paper/6XH2FXNA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.22925&json=true","fetch_graph":"https://pith.science/api/pith-number/6XH2FXNA7LA3WU2RHBSRT5CRW7/graph.json","fetch_events":"https://pith.science/api/pith-number/6XH2FXNA7LA3WU2RHBSRT5CRW7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7/action/storage_attestation","attest_author":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7/action/author_attestation","sign_citation":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7/action/citation_signature","submit_replication":"https://pith.science/pith/6XH2FXNA7LA3WU2RHBSRT5CRW7/action/replication_record"}},"created_at":"2026-07-28T00:22:02.403843+00:00","updated_at":"2026-07-28T00:22:02.403843+00:00"}