{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:IW3VIUTTAZXJV5WNFJ5SN5FY4U","short_pith_number":"pith:IW3VIUTT","canonical_record":{"source":{"id":"2602.21947","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-02-25T14:32:15Z","cross_cats_sorted":[],"title_canon_sha256":"438db4d5b341362553dfc592b579c836629a2fb5d41bd991285198e03575af59","abstract_canon_sha256":"8cba68210066087d5ea0ad0b49dae0ff41901c5d32fee6cfe3046bab57a2a2f1"},"schema_version":"1.0"},"canonical_sha256":"45b7545273066e9af6cd2a7b26f4b8e50fded2c25b5620de1dd6f293096be22c","source":{"kind":"arxiv","id":"2602.21947","version":5},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2602.21947","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"arxiv_version","alias_value":"2602.21947v5","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.21947","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"pith_short_12","alias_value":"IW3VIUTTAZXJ","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"pith_short_16","alias_value":"IW3VIUTTAZXJV5WN","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"pith_short_8","alias_value":"IW3VIUTT","created_at":"2026-07-28T01:22:30Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:IW3VIUTTAZXJV5WNFJ5SN5FY4U","target":"record","payload":{"canonical_record":{"source":{"id":"2602.21947","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-02-25T14:32:15Z","cross_cats_sorted":[],"title_canon_sha256":"438db4d5b341362553dfc592b579c836629a2fb5d41bd991285198e03575af59","abstract_canon_sha256":"8cba68210066087d5ea0ad0b49dae0ff41901c5d32fee6cfe3046bab57a2a2f1"},"schema_version":"1.0"},"canonical_sha256":"45b7545273066e9af6cd2a7b26f4b8e50fded2c25b5620de1dd6f293096be22c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T01:22:30.389423Z","signature_b64":"lqQPO2F9qTWhjqWGCaeerS/iQgcTPr7yVHckAkBE2T4+/RL66zYX+9tkZ3p8pD25OGKS9UEQ4w1zSZds1TDqDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45b7545273066e9af6cd2a7b26f4b8e50fded2c25b5620de1dd6f293096be22c","last_reissued_at":"2026-07-28T01:22:30.388470Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T01:22:30.388470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2602.21947","source_version":5,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-28T01:22:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"wIvFQLUefxyyCpC5ezZ0ru+c3+BHr2Qdtug3moMPQGdwypdyc9GUxQzPzbD87i6h/P48Mwge+/CCmiIx/KKtDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T11:17:20.541170Z"},"content_sha256":"1af39d22c12d0225800e7cc6bad903b40b6921ace30e14029c9cc4aa44e8e38a","schema_version":"1.0","event_id":"sha256:1af39d22c12d0225800e7cc6bad903b40b6921ace30e14029c9cc4aa44e8e38a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:IW3VIUTTAZXJV5WNFJ5SN5FY4U","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Algorithmic Blindness in Large Language Models: A Calibration Study of Performance Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Large language models exhibit systematic failure at predicting algorithmic outcomes despite knowing algorithm facts.","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ashish Mahendran Kurapath, Sohan Venkatesh, Tejas Melkote","submitted_at":"2026-02-25T14:32:15Z","abstract_excerpt":"Large language models (LLMs) demonstrate remarkable breadth of knowledge, yet their ability to reason about computational processes remains poorly understood. Closing this gap matters for practitioners who rely on LLMs to guide algorithm selection and deployment. We address this limitation using causal discovery as a testbed and evaluate eight frontier LLMs against ground truth derived from algorithm executions. We find systematic, near-total failure across models. The predicted ranges are far wider than true confidence intervals yet still fail to contain the true algorithmic mean in most case"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We find systematic, near-total failure across models. The predicted ranges are far wider than true confidence intervals yet still fail to contain the true algorithmic mean in most cases. Most models perform worse than random guessing and the best model's marginal improvement is attributable to benchmark memorization rather than principled reasoning.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That performance on causal discovery tasks constitutes a valid and generalizable test of algorithmic reasoning ability rather than a narrow or confounded proxy.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"LLMs exhibit algorithmic blindness, producing predictions whose ranges miss true algorithmic means in most cases and often perform worse than random guessing.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Large language models exhibit systematic failure at predicting algorithmic outcomes despite knowing algorithm facts.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"87d2bd44e8b70df5dd228f7cdc7d9ce9e6bd1c12c15ef46dab28ab2256bb6a8d"},"source":{"id":"2602.21947","kind":"arxiv","version":5},"verdict":{"id":"d8592059-2f15-42c9-a21b-ad06e4eede8d","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-15T19:36:05.513234Z","strongest_claim":"We find systematic, near-total failure across models. The predicted ranges are far wider than true confidence intervals yet still fail to contain the true algorithmic mean in most cases. Most models perform worse than random guessing and the best model's marginal improvement is attributable to benchmark memorization rather than principled reasoning.","one_line_summary":"LLMs exhibit algorithmic blindness, producing predictions whose ranges miss true algorithmic means in most cases and often perform worse than random guessing.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That performance on causal discovery tasks constitutes a valid and generalizable test of algorithmic reasoning ability rather than a narrow or confounded proxy.","pith_extraction_headline":"Large language models exhibit systematic failure at predicting algorithmic outcomes despite knowing algorithm facts."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.21947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"d8592059-2f15-42c9-a21b-ad06e4eede8d"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-28T01:22:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+sX+sq/CqbpCGRRdWY1ypY8xsCCerYzwm3pzNKM9fa1TPwCUWvekz/hTKMRrovYZAuivrv8OETrupSKrCYeADw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T11:17:20.541847Z"},"content_sha256":"bb7c645c38dd47e6277a8483b065b5b2f3e6f68281cfdce041bf14607ba193d0","schema_version":"1.0","event_id":"sha256:bb7c645c38dd47e6277a8483b065b5b2f3e6f68281cfdce041bf14607ba193d0"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/IW3VIUTTAZXJV5WNFJ5SN5FY4U/bundle.json","state_url":"https://pith.science/pith/IW3VIUTTAZXJV5WNFJ5SN5FY4U/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/IW3VIUTTAZXJV5WNFJ5SN5FY4U/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T11:17:20Z","links":{"resolver":"https://pith.science/pith/IW3VIUTTAZXJV5WNFJ5SN5FY4U","bundle":"https://pith.science/pith/IW3VIUTTAZXJV5WNFJ5SN5FY4U/bundle.json","state":"https://pith.science/pith/IW3VIUTTAZXJV5WNFJ5SN5FY4U/state.json","well_known_bundle":"https://pith.science/.well-known/pith/IW3VIUTTAZXJV5WNFJ5SN5FY4U/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:IW3VIUTTAZXJV5WNFJ5SN5FY4U","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8cba68210066087d5ea0ad0b49dae0ff41901c5d32fee6cfe3046bab57a2a2f1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-02-25T14:32:15Z","title_canon_sha256":"438db4d5b341362553dfc592b579c836629a2fb5d41bd991285198e03575af59"},"schema_version":"1.0","source":{"id":"2602.21947","kind":"arxiv","version":5}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2602.21947","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"arxiv_version","alias_value":"2602.21947v5","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.21947","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"pith_short_12","alias_value":"IW3VIUTTAZXJ","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"pith_short_16","alias_value":"IW3VIUTTAZXJV5WN","created_at":"2026-07-28T01:22:30Z"},{"alias_kind":"pith_short_8","alias_value":"IW3VIUTT","created_at":"2026-07-28T01:22:30Z"}],"graph_snapshots":[{"event_id":"sha256:bb7c645c38dd47e6277a8483b065b5b2f3e6f68281cfdce041bf14607ba193d0","target":"graph","created_at":"2026-07-28T01:22:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"We find systematic, near-total failure across models. The predicted ranges are far wider than true confidence intervals yet still fail to contain the true algorithmic mean in most cases. Most models perform worse than random guessing and the best model's marginal improvement is attributable to benchmark memorization rather than principled reasoning."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That performance on causal discovery tasks constitutes a valid and generalizable test of algorithmic reasoning ability rather than a narrow or confounded proxy."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"LLMs exhibit algorithmic blindness, producing predictions whose ranges miss true algorithmic means in most cases and often perform worse than random guessing."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Large language models exhibit systematic failure at predicting algorithmic outcomes despite knowing algorithm facts."}],"snapshot_sha256":"87d2bd44e8b70df5dd228f7cdc7d9ce9e6bd1c12c15ef46dab28ab2256bb6a8d"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2602.21947/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models (LLMs) demonstrate remarkable breadth of knowledge, yet their ability to reason about computational processes remains poorly understood. Closing this gap matters for practitioners who rely on LLMs to guide algorithm selection and deployment. We address this limitation using causal discovery as a testbed and evaluate eight frontier LLMs against ground truth derived from algorithm executions. We find systematic, near-total failure across models. The predicted ranges are far wider than true confidence intervals yet still fail to contain the true algorithmic mean in most case","authors_text":"Ashish Mahendran Kurapath, Sohan Venkatesh, Tejas Melkote","cross_cats":[],"headline":"Large language models exhibit systematic failure at predicting algorithmic outcomes despite knowing algorithm facts.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-02-25T14:32:15Z","title":"Algorithmic Blindness in Large Language Models: A Calibration Study of Performance Prediction"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.21947","kind":"arxiv","version":5},"verdict":{"created_at":"2026-05-15T19:36:05.513234Z","id":"d8592059-2f15-42c9-a21b-ad06e4eede8d","model_set":{"reader":"grok-4.3"},"one_line_summary":"LLMs exhibit algorithmic blindness, producing predictions whose ranges miss true algorithmic means in most cases and often perform worse than random guessing.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Large language models exhibit systematic failure at predicting algorithmic outcomes despite knowing algorithm facts.","strongest_claim":"We find systematic, near-total failure across models. The predicted ranges are far wider than true confidence intervals yet still fail to contain the true algorithmic mean in most cases. Most models perform worse than random guessing and the best model's marginal improvement is attributable to benchmark memorization rather than principled reasoning.","weakest_assumption":"That performance on causal discovery tasks constitutes a valid and generalizable test of algorithmic reasoning ability rather than a narrow or confounded proxy."}},"verdict_id":"d8592059-2f15-42c9-a21b-ad06e4eede8d"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1af39d22c12d0225800e7cc6bad903b40b6921ace30e14029c9cc4aa44e8e38a","target":"record","created_at":"2026-07-28T01:22:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8cba68210066087d5ea0ad0b49dae0ff41901c5d32fee6cfe3046bab57a2a2f1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-02-25T14:32:15Z","title_canon_sha256":"438db4d5b341362553dfc592b579c836629a2fb5d41bd991285198e03575af59"},"schema_version":"1.0","source":{"id":"2602.21947","kind":"arxiv","version":5}},"canonical_sha256":"45b7545273066e9af6cd2a7b26f4b8e50fded2c25b5620de1dd6f293096be22c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"45b7545273066e9af6cd2a7b26f4b8e50fded2c25b5620de1dd6f293096be22c","first_computed_at":"2026-07-28T01:22:30.388470Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-28T01:22:30.388470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"lqQPO2F9qTWhjqWGCaeerS/iQgcTPr7yVHckAkBE2T4+/RL66zYX+9tkZ3p8pD25OGKS9UEQ4w1zSZds1TDqDA==","signature_status":"signed_v1","signed_at":"2026-07-28T01:22:30.389423Z","signed_message":"canonical_sha256_bytes"},"source_id":"2602.21947","source_kind":"arxiv","source_version":5}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1af39d22c12d0225800e7cc6bad903b40b6921ace30e14029c9cc4aa44e8e38a","sha256:bb7c645c38dd47e6277a8483b065b5b2f3e6f68281cfdce041bf14607ba193d0"],"state_sha256":"e7d7997327a681cb044d118adc4bc3541e264dab89aaf738a21e1afa77e1cf01"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mgJxJH9zrtVA8O8/OZZ6ZM/yxa3ITU1te8w6qaPNLWV7CmyV2vzlwC7obvpCi2wGKmtyXYaBgBoDEQOhsridCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T11:17:20.547428Z","bundle_sha256":"a09c40ce3157de64906c616ce8f0c04205eda8fd2e455edc3e11f6b22673e816"}}