{"as_of":"2026-08-08T05:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:12f59cc9f722b6d71c81e35f604925998afba6ccf69a3578a674bb0e106fbfe1","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:29:03.805548Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-03T21:20:00.041277Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T21:28:58.386085Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2605.05737","last_updated":"2026-05-07T06:29:34Z","snapshot_observed_at":"2026-07-06T23:18:17.333660Z","submitted_at":"2026-05-07T06:29:34Z","title":"ReFlect: An Effective Harness System for Complex Long-Horizon LLM Reasoning","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-08T11:35:25.204350Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2605.05737"},"observation_digest":"sha256:c02bb7ece9bc5dabb2b9bb792b15b0415493ed1121920f7820812a39862cf055","observation_id":"3547e909-d01a-42bc-8fbe-10036c058fef","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2605.20867","last_updated":"2026-05-20T08:02:15Z","snapshot_observed_at":"2026-07-06T23:31:23.407780Z","submitted_at":"2026-05-20T08:02:15Z","title":"ProCrit: Self-Elicited Multi-Perspective Reasoning with Critic-Guided Revision for Multimodal Sarcasm Detection","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T02:35:37.724212Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2605.20867"},"observation_digest":"sha256:0782dd64adbb782b598fa26c8a36e070a615abf56fceaac1df91e65137c8da42","observation_id":"96d7ffc6-6dae-45a3-989d-37e0f37d0836","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2606.29425","last_updated":"2026-06-28T14:40:01Z","snapshot_observed_at":"2026-08-05T16:29:32.428034Z","submitted_at":"2026-06-28T14:40:01Z","title":"Mixture of Debaters: Learn to Debate at Architectural Level in Multi-Agent Reasoning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-30T07:11:02.464556Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2606.29425"},"observation_digest":"sha256:92f905919db9965d3fd1ddcdc1efb75d019e1b209e5396945f95d047b2e9aca4","observation_id":"3a91d3a0-6f38-4726-acc8-06b923d8d159","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"cited_work":{"arxiv_id":"2507.02778","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02778","snapshot_observed_at":"2026-08-04T02:29:36.375484Z","title":"Self-correction bench: Uncovering and addressing the self-correction blind spot in large language models","venue":null,"work_id":"cdc79fb8-0919-4089-8a57-ab9adbd958d7","year":2025},"citing_paper":{"arxiv_id":"2607.02089","last_updated":"2026-07-01T14:25:43Z","snapshot_observed_at":"2026-08-02T16:57:02.105465Z","submitted_at":"2026-07-01T14:25:43Z","title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-07-03T21:20:00.041277Z"},"links":{"cited_paper":"/paper/2507.02778","citing_paper":"/paper/2607.02089"},"observation_digest":"sha256:2ec9a0a140f5fc2c6f43e96e168baa137bc806d746f115d5b5adf70be6f99595","observation_id":"eaa9921f-da88-4f3c-9369-e2f5a7f177a6","resolution":{"observed_at":"2026-08-04T02:29:36.375484Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.02778/citation-record","integrity":"/paper/2507.02778/integrity","json":"/paper/2507.02778/citation-record.json","paper":"/paper/2507.02778"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T20:28:55.823384Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:55.823384Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:bdf600a5e9bbdf725f39810ec333c1d5351350048af2e9f353e49f465b4b383a","observation_id":"19a388be-1abf-4a33-ba82-7b35e40fd82f","resolution":{"observed_at":"2026-08-06T20:28:55.823384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:08.675442Z","title":"The claude 3 model family: Opus, sonnet, haiku, Mar 2024","venue":null,"work_id":"a1f4a303-4080-4221-8ca8-887d35541a91","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:55.944606Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:3025982ffc81acb24b9b4d0cdab37a32df3512e1d2ec64574b7d3a9fcd1f329e","observation_id":"73f0cd9a-0b1b-408e-8072-9d94b04998e2","resolution":{"observed_at":"2026-08-06T20:29:08.781161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:08.375109Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities., June 2025","venue":null,"work_id":"74706def-875a-4674-8807-e696c5867219","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.097504Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:1c1ad316ea790da524c76b63b035376944396f09c13a97ba8753fddb06dc1769","observation_id":"50bbfc04-ce8c-4b4b-b5b0-3fff3e8b3f5a","resolution":{"observed_at":"2026-08-06T20:29:08.531479Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T20:28:56.227672Z","title":"Qwen3 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.227672Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:a0a8f6530b3be77859e622d0dfca0274100f57da6327829168a69ae44fb4bf6f","observation_id":"ae1989c8-73a6-49a0-8fc4-2d16699ff84d","resolution":{"observed_at":"2026-08-06T20:28:56.227672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:08.150401Z","title":"The llama 4 herd: The beginning of a new era of natively multimodal ai innovation, Apr 2025","venue":null,"work_id":"5b319c6b-8bd2-4a91-80bd-babcacb6c457","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.333821Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:8370a31e980ad09314f9ed60933d767629a6dba6eb1b60b82383a94f4bc87ef5","observation_id":"586457e5-ab3b-4b4a-9464-8e20506298a9","resolution":{"observed_at":"2026-08-06T20:29:08.248924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T20:28:56.531856Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.531856Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:44eddd690547449b12eaacce63ae713ad9ff1172c1c3c9bcef9d93aced5566f2","observation_id":"91ab01f4-557c-433b-b672-94dbef010bc0","resolution":{"observed_at":"2026-08-06T20:28:56.531856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:56.646944Z","title":"On faithfulness and factuality in abstractive summarization","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.646944Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:0323ba82fd3790475a427b73696e3ea2385b04345c49ff480f56ecafa90f968f","observation_id":"f38e80ac-bd0f-42de-b022-b8b7c6500e47","resolution":{"observed_at":"2026-08-06T20:28:56.646944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:56.783467Z","title":"A survey on hallucination in large language models: Principles, taxonomy, challenges, and open questions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.783467Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:5c94dad506ef5959d8c2de17f3f7352a725076a3882015cd6cb69fdb13ffd09f","observation_id":"77338fbe-e90f-4bc5-bba8-76b029fc8ea4","resolution":{"observed_at":"2026-08-06T20:28:56.783467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:56.946027Z","title":"Do, Yan Xu, and Pascale Fung","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:56.946027Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:824c5bb47fd1ea20f5f81fbfa0116ffaf94cb908eaffcb67b95dca46dded8240","observation_id":"098af2f9-edf9-49c1-a735-ebf863240b3d","resolution":{"observed_at":"2026-08-06T20:28:56.946027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.071428Z","title":"Large language models can be easily distracted by irrelevant context","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.071428Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:a9ea0184110478786610723ebbc8e18b3fe1caf62b003a570f7e97e4d8acdc39","observation_id":"caa6fbd6-6cbc-4f13-b88a-1bcb810818e9","resolution":{"observed_at":"2026-08-06T20:28:57.071428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02061","last_updated":"2025-03-05T01:58:08Z","snapshot_observed_at":"2026-08-07T22:23:28.372208Z","submitted_at":"2024-06-04T07:43:33Z","title":"Alice in Wonderland: Simple Tasks Showing Complete Reasoning Breakdown in State-Of-the-Art Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02061","snapshot_observed_at":"2026-08-06T20:28:57.217863Z","title":"Alice in wonderland: Simple tasks showing complete reasoning breakdown in state-of-the-art large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.217863Z"},"links":{"cited_paper":"/paper/2406.02061","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:2562f9934a2a4f5d27db629777e699b934eda6c98db78a4889a3245df9caae07","observation_id":"d87bcd25-bf69-48cd-b7fa-c14ce7548102","resolution":{"observed_at":"2026-08-06T20:28:57.217863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.881955Z","title":"Reflexion: language agents with verbal reinforcement learning","venue":null,"work_id":"ec724c1c-233d-4a90-b6ef-e04a5ac88a3a","year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.332058Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:7531474b8321e89b393cf6ee555418e0ad4e9208f7be5674ddc05dd04793b486","observation_id":"fae702c4-d4da-4075-8c5e-a0238f612dfe","resolution":{"observed_at":"2026-08-06T20:29:08.005937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.429687Z","title":"Self-refine: Iterative refinement with self-feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.429687Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:cacd72258b0e19cfbc66a23f45164cd6155fcc29f6ebb43bfb4660b16478256d","observation_id":"1cecc2d2-7db8-48a5-94cf-cff9ceed325b","resolution":{"observed_at":"2026-08-06T20:28:57.429687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.590435Z","title":"Language models can solve computer tasks","venue":null,"work_id":"a220d232-22b8-45b3-91bd-ee74e3bc794f","year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.560434Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:3204ea306da5f071dad9d59733a673fc6f1f897785c71c601c41f26f81c006c9","observation_id":"0952229d-8e2e-4192-9c73-59e1a4b6dcff","resolution":{"observed_at":"2026-08-06T20:29:07.724534Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.717223Z","title":"When can LLM s actually correct their own mistakes? a critical survey of self-correction of LLM s","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.717223Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:4f2830a93c6552df60376725b5e297b2dd01a6a70c5381a21395ce3fa223d746","observation_id":"ee98ddf2-8acf-4313-9d25-4c6955fe54ee","resolution":{"observed_at":"2026-08-06T20:28:57.717223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01798","last_updated":"2024-03-14T04:27:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-03T04:56:12Z","title":"Large Language Models Cannot Self-Correct Reasoning Yet","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01798","snapshot_observed_at":"2026-08-06T20:28:57.838761Z","title":"Large language models cannot self-correct reasoning yet, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.838761Z"},"links":{"cited_paper":"/paper/2310.01798","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:ae0df6bd87395211f474b56e313df0c88c1ff66244645f9641ed17b3ac92d9eb","observation_id":"7cb640df-e27c-4882-80c9-919eeba976dc","resolution":{"observed_at":"2026-08-06T20:28:57.838761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:57.998432Z","title":"LLM s cannot find reasoning errors, but can correct them given the error location","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:57.998432Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:1c1f4372530925dbe12cc850675b8f57bd32411522460a60d7e7eed87af88153","observation_id":"8923266a-b790-4707-a354-bce566c6368d","resolution":{"observed_at":"2026-08-06T20:28:57.998432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.333882Z","title":"Evaluating LLM s at detecting errors in LLM responses","venue":null,"work_id":"2cb19ed8-ba5b-43ad-81e8-940b0d113aa5","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.142786Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:4832861b4be4f83f1471d2b250838fc0bb06f103ee5fb95f4bf11b08e2d4da87","observation_id":"ed5bf506-a601-41b5-a800-544053498431","resolution":{"observed_at":"2026-08-06T20:29:07.454677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:07.103360Z","title":"Training language models to self-correct via reinforcement learning","venue":null,"work_id":"00012590-cccf-4ffb-901d-43484ac3ffa2","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.292638Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:2078769dc19447d48e2982e9746f6ea6244fb69a39d9f6594971c2a725682d1d","observation_id":"5f1f55b3-3870-468a-9f67-1af9ce35a562","resolution":{"observed_at":"2026-08-06T20:29:07.214068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:06.774596Z","title":"Jailbroken: how does llm safety training fail? In Proceedings of the 37th International Conference on Neural Information Processing Systems, NIPS '23, Red Hook, NY, USA, 2023","venue":null,"work_id":"21c4619b-07e3-4ff7-b6be-600055be1f44","year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.454749Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:9453588348f0992c8565b6832535f7d510bf8e9dfdf623dbc34cf8ffbdff93a3","observation_id":"efddf2d7-1f69-4927-bb0a-f4f3896e609c","resolution":{"observed_at":"2026-08-06T20:29:06.943034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:06.452420Z","title":"Formalizing and benchmarking prompt injection attacks and defenses","venue":null,"work_id":"34c5eb3d-fefe-4d14-b227-e27febc8ec52","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.558097Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:434370fd6888258efd531626ff1e39343caec3019b8be4fd84029f9bd134dd81","observation_id":"553309ad-0b0f-4417-b8b1-f864b3625a67","resolution":{"observed_at":"2026-08-06T20:29:06.592654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13702","last_updated":"2023-07-17T01:08:39Z","snapshot_observed_at":"2026-07-31T04:21:27.501709Z","submitted_at":"2023-07-17T01:08:39Z","title":"Measuring Faithfulness in Chain-of-Thought Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13702","snapshot_observed_at":"2026-08-06T20:28:58.704896Z","title":"Bowman, and Ethan Perez","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.704896Z"},"links":{"cited_paper":"/paper/2307.13702","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:bbb68cad7e465828e9ba0fa270a369c2c65a3f17da0190e5763ac2782029d0d2","observation_id":"7caeb070-8c3f-4b21-a32e-ab3575a5ea7d","resolution":{"observed_at":"2026-08-06T20:28:58.704896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:06.139227Z","title":null,"venue":null,"work_id":"0144abe0-92f2-4653-9058-40959eaac168","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:58.869328Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:24d235d111dffd7349476165d674c53328607795b047c1b23770d121eafb8908","observation_id":"9aced9d1-1d08-48f8-87f4-1e5b8e6a1a5b","resolution":{"observed_at":"2026-08-06T20:29:06.302848Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.074923Z","title":"Scaling LLM test-time compute optimally can be more effective than scaling parameters for reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.074923Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:ee85035a99e96adea9494c3c18da278e2d8994a560901db6df4c1012337af919","observation_id":"6e96adf5-08a3-472a-8d93-9315cfd851d5","resolution":{"observed_at":"2026-08-06T20:28:59.074923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19393","last_updated":"2025-03-01T06:07:39Z","snapshot_observed_at":"2026-07-06T20:29:11.710285Z","submitted_at":"2025-01-31T18:48:08Z","title":"s1: Simple test-time scaling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19393","snapshot_observed_at":"2026-08-06T20:28:59.270560Z","title":"s1: Simple test-time scaling, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.270560Z"},"links":{"cited_paper":"/paper/2501.19393","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6caee72957f19e16b84c37747bc599323b820a19f7aebe16a65dfc6115008b12","observation_id":"a61249e0-c3ef-455b-b871-6c8f7c79c70c","resolution":{"observed_at":"2026-08-06T20:28:59.270560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.400211Z","title":"Benchmarking cognitive biases in large language models as evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.400211Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:0b9d454d297717dc470b57fe3c180e7838c337f37ada022c4687d96a0a061b64","observation_id":"f0954cbd-2c14-4b6b-8801-899f2a583ed1","resolution":{"observed_at":"2026-08-06T20:28:59.400211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.570749Z","title":"Cognitive bias in decision-making with LLM s","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.570749Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:b33d2b0b0e65c1a64315adda3535d5625588f9006356822ca144fbde9b3a5c25","observation_id":"78c11968-cc45-47a7-b086-2d93729aa853","resolution":{"observed_at":"2026-08-06T20:28:59.570749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:05.897396Z","title":"Capturing failures of large language models via human cognitive biases","venue":null,"work_id":"ef366fa0-dba3-4fd4-b94d-f026b5b698f1","year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.745163Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:e6f39f8900bc1aeac31fb4dc7454fb470506805cfadf4a852f565e26d403ae81","observation_id":"bc521204-141d-47ff-be90-5f34ec347f6e","resolution":{"observed_at":"2026-08-06T20:29:06.008527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:28:59.906202Z","title":"Lin, and Lee Ross","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T20:28:59.906202Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:39db37328ff10f6c621b9ac35925227ff7420cf5fe54c9e22620aca2d2f59c96","observation_id":"42e7bac0-e04d-4130-8fc7-375e883a226c","resolution":{"observed_at":"2026-08-06T20:28:59.906202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06559","last_updated":"2025-05-26T14:03:32Z","snapshot_observed_at":"2026-08-05T20:30:49.812919Z","submitted_at":"2024-12-09T15:11:40Z","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06559","snapshot_observed_at":"2026-08-06T20:29:00.072960Z","title":"Processbench: Identifying process errors in mathematical reasoning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.072960Z"},"links":{"cited_paper":"/paper/2412.06559","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6c40740dac64b1d12f9e7121727457d2d4bec21aa455abb6f839ff18654cfdd2","observation_id":"2c893750-911c-4989-a3ab-7264afd41c4f","resolution":{"observed_at":"2026-08-06T20:29:00.072960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03124","last_updated":"2025-06-28T12:31:45Z","snapshot_observed_at":"2026-08-07T08:05:38.374743Z","submitted_at":"2025-01-06T16:31:45Z","title":"PRMBench: A Fine-grained and Challenging Benchmark for Process-Level Reward Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03124","snapshot_observed_at":"2026-08-06T20:29:00.154021Z","title":"Prmbench: A fine-grained and challenging benchmark for process-level reward models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.154021Z"},"links":{"cited_paper":"/paper/2501.03124","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:796ca3232038a9f82563c37aa1b4bb00e0dc2a3983bfe9b44d38221e855c5da3","observation_id":"af15ffc1-4b55-4d7b-893f-55462028cbe3","resolution":{"observed_at":"2026-08-06T20:29:00.154021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T20:29:00.206673Z","title":"Training verifiers to solve math word problems, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.206673Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:e00f64d94833b3feb4222a0c0c6925707ceed22421381a2c103d25ca92f31c11","observation_id":"242c2d42-a346-4983-a5f9-e9fb052d06a4","resolution":{"observed_at":"2026-08-06T20:29:00.206673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:05.606335Z","title":"Introducing gpt-4.1 in the api, Apr 2025","venue":null,"work_id":"f30196ea-2d3d-4c0b-9be8-609d76c9ffc5","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.270483Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:3b793d0487e2a1f3f1d05bc173458f29b2f617caf9d4ee3c998b8cb7d748c8e3","observation_id":"0a682beb-9d8b-4ffc-8185-1c377d0db050","resolution":{"observed_at":"2026-08-06T20:29:05.741704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:00.355018Z","title":"Let's verify step by step","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.355018Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:5f1801ddcfab59147716300d4d17bd0d744ca63f5403f96c21211341dd939131","observation_id":"547769ab-98dd-42f4-8e5d-f44fecc985c4","resolution":{"observed_at":"2026-08-06T20:29:00.355018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:00.438344Z","title":"Measuring mathematical problem solving with the MATH dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.438344Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:deddd798fee44866379c30824ac6f5c0e0805e23dc1371d7d9602259fc73306e","observation_id":"cdac7ea2-3ff3-405a-9392-f1aa43da387b","resolution":{"observed_at":"2026-08-06T20:29:00.438344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:00.521903Z","title":"Transformers: State-of-the-art natural language processing","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.521903Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:73095f91931d447dfa80c9924c787f5093814b7c96296b73750bed83e17fa8dc","observation_id":"d1a30e71-b262-458f-8efa-c396237e35b6","resolution":{"observed_at":"2026-08-06T20:29:00.521903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T20:29:00.633907Z","title":"Zhang, Han Bao, Hanwei Xu, Haocheng Wang, Haowei Zhang, Honghui Ding, Huajian Xin, Huazuo Gao, Hui Li, Hui Qu, J","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.633907Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:a96a5a97578bcc774f60e50bf4afb1f7138591920d032475269375877fe87f26","observation_id":"98372d2d-19f9-45e5-8238-be050bbd90ed","resolution":{"observed_at":"2026-08-06T20:29:00.633907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T20:29:00.711530Z","title":"Qwen2.5 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.711530Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:3d70ae7e32c8bac72ea2fa3f229959351ed69dc54f24a9d1cc4d26762f5bd77f","observation_id":"5e52510a-4855-4949-9e1d-a50fd817f0f4","resolution":{"observed_at":"2026-08-06T20:29:00.711530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:05.285132Z","title":"Llama 3.3, Dec 2024","venue":null,"work_id":"560e69f3-3bad-4c9b-a900-f1bcb8599db2","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.790776Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:4f32c97880dde1a435cb3023556d0bd1eec3ec58b2db289cebfcc160f1f9715b","observation_id":"a5b12f90-c1f5-4320-988b-69bde06197c9","resolution":{"observed_at":"2026-08-06T20:29:05.454473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-05T04:04:21.846023Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-06T20:29:00.903257Z","title":"Hewett, Mojan Javaheripi, Piero Kauffmann, James R","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:00.903257Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:6f284ae9831fbad5b01caea62863abf4d90ced14e54667d5e96193da0004a3a8","observation_id":"070d8f16-3cc9-4caa-ba86-afcde95ccfc4","resolution":{"observed_at":"2026-08-06T20:29:00.903257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-06T20:29:01.073515Z","title":"Qwen2 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.073515Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:319c80c6d0511cf68a29dd29f83d87892360ccdd7c5d2312ad2d739e86500f19","observation_id":"d8898eac-0793-4381-a9b0-2df54c6e3a79","resolution":{"observed_at":"2026-08-06T20:29:01.073515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T20:29:01.235737Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.235737Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:2cc8d7bc77512073b46e234323883ec16b7c6a08d8166c9e9c3023f8b98b01b1","observation_id":"598889cd-4a56-43db-8cd4-7e4e3b4c55bc","resolution":{"observed_at":"2026-08-06T20:29:01.235737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:04.911354Z","title":"Mistral small 3, Jan 2025","venue":null,"work_id":"a179b96e-d5d0-4075-9948-7e1d2bd6577e","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.386989Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:8394cfe3db6bc5e7de71e363c5012e974b329090ab34933267f60b99e69b44c4","observation_id":"9c5dc13c-19b8-4c72-8e21-1ba13f577e88","resolution":{"observed_at":"2026-08-06T20:29:05.122636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-06T20:29:01.501958Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.501958Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:26bfd41365fdf1e5a831d9f5b9e921720d2ab9adc4e1c5b6eefa2987195836f1","observation_id":"91a1453a-6c9b-4155-aebf-55fa2131cd9f","resolution":{"observed_at":"2026-08-06T20:29:01.501958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:01.660462Z","title":"Impact of pretraining term frequencies on few-shot numerical reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.660462Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:1875b538a9da3e07c1391c6f618cf25cfb3e45601d2a7b3e8801c25f41da3001","observation_id":"8a78e84b-9451-4307-9a7a-d78784925c50","resolution":{"observed_at":"2026-08-06T20:29:01.660462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:04.670597Z","title":"Smith, Sarah Wiegreffe, and Yanai Elazar","venue":null,"work_id":"31a41ed1-6737-45ae-ac97-daf971976ed6","year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.845245Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:5182b2067b1d27b6926a2eca05363d71a89966b204d131ff6c4bc20db5253115","observation_id":"a8943ef9-1315-4cd0-8398-f439535fbb8e","resolution":{"observed_at":"2026-08-06T20:29:04.782194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:01.991836Z","title":"o pf, Yannic Kilcher, Dimitri von R \\","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:01.991836Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:9c9351daa39af124c4b1038f96161907881e342a03a9e9b7f952b080256f225a","observation_id":"e5ca97be-2dab-4115-a7b7-171ee449086e","resolution":{"observed_at":"2026-08-06T20:29:01.991836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.140460Z","title":"Openhermes 2.5: An open dataset of synthetic data for generalist llm assistants, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.140460Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:afd76aca1f101d77c37e2790123c988f02e4de8f538e08e4439da04cd0b5396e","observation_id":"9472acb6-eb08-412e-a62b-0d21af73a4f9","resolution":{"observed_at":"2026-08-06T20:29:02.140460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11116","last_updated":"2025-06-09T06:37:15Z","snapshot_observed_at":"2026-08-07T05:30:36.959314Z","submitted_at":"2025-06-09T06:37:15Z","title":"Infinity Instruct: Scaling Instruction Selection and Synthesis to Enhance Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11116","snapshot_observed_at":"2026-08-06T20:29:02.244807Z","title":"Infinity instruct: Scaling instruction selection and synthesis to enhance language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.244807Z"},"links":{"cited_paper":"/paper/2506.11116","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:711fb994ffd591129ce73eb60a6142b77aa79c2f5c6c99cc969e2e989bbafb2a","observation_id":"c317eade-a7e4-48ad-8654-5687149260d8","resolution":{"observed_at":"2026-08-06T20:29:02.244807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:04.405551Z","title":"Ultrafeedback: Boosting language models with high-quality feedback, 2024","venue":null,"work_id":"63259950-d6ec-4139-bdd8-68e6fe16695a","year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.356141Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:c3d22ca3f5f3ed7d961ca0dba85ea3559d6ffdbca622631c3198e488e93a49a1","observation_id":"61ed6acb-e56a-49e0-8e0a-49c8c23c40b5","resolution":{"observed_at":"2026-08-06T20:29:04.501080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.487929Z","title":"Hwang, Jiangjiang Yang, Ronan Le Bras, Oyvind Tafjord, Christopher Wilhelm, Luca Soldaini, Noah A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.487929Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:7d4a2e108fe96be94f0f126067508ecb9880a62c99da081849535c26910e6ae6","observation_id":"d42f5cd2-00f6-4eeb-98c0-831a16acdeb7","resolution":{"observed_at":"2026-08-06T20:29:02.487929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.599732Z","title":"Open r1: A fully open reproduction of deepseek-r1, January 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.599732Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:e3023db1bda9928ae2fc817338e54972da7a75489e5658cf8cfd7a7601c8dbc2","observation_id":"fa7a74bf-6fcc-4198-83eb-b128ade148e9","resolution":{"observed_at":"2026-08-06T20:29:02.599732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04178","last_updated":"2025-06-05T02:21:52Z","snapshot_observed_at":"2026-08-03T02:41:13.331563Z","submitted_at":"2025-06-04T17:25:39Z","title":"OpenThoughts: Data Recipes for Reasoning Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.04178","snapshot_observed_at":"2026-08-06T20:29:02.783626Z","title":"Merrill, Tatsunori Hashimoto, Yejin Choi, Jenia Jitsev, Reinhard Heckel, Maheswaran Sathiamoorthy, Alexandros G","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.783626Z"},"links":{"cited_paper":"/paper/2506.04178","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:e33eed2e474efda4bf3c86439f5dbafeba851ca76a0754d5fedce3c346535630","observation_id":"fab9b386-0834-49e7-9db5-dc727f9131ea","resolution":{"observed_at":"2026-08-06T20:29:02.783626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:02.938680Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:02.938680Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:dfa6bb9057b25478f898f5078c1577e3a710ab5edbf14c794826c919942c9557","observation_id":"432e34ca-6d12-4311-9df2-c0977733d550","resolution":{"observed_at":"2026-08-06T20:29:02.938680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.20689","last_updated":"2024-03-29T07:17:39Z","snapshot_observed_at":"2026-08-07T10:31:33.290602Z","submitted_at":"2023-10-31T17:52:22Z","title":"Learning From Mistakes Makes LLM Better Reasoner","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.20689","snapshot_observed_at":"2026-08-06T20:29:03.095630Z","title":"Learning from mistakes makes llm better reasoner, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.095630Z"},"links":{"cited_paper":"/paper/2310.20689","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:bf24694ae0056f9294c2bf5d81284b42ef3e8e123af0bc6cc95a285e1bf421f7","observation_id":"5a580e71-de61-4a51-819e-81f6b955611e","resolution":{"observed_at":"2026-08-06T20:29:03.095630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17703","last_updated":"2025-03-29T15:21:55Z","snapshot_observed_at":"2026-08-06T17:25:43.987350Z","submitted_at":"2025-01-29T15:20:30Z","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17703","snapshot_observed_at":"2026-08-06T20:29:03.216236Z","title":"Critique fine-tuning: Learning to critique is more effective than learning to imitate, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.216236Z"},"links":{"cited_paper":"/paper/2501.17703","citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:544cc1133426d934c517dc56345cc9ab158dc3ac4546225ed20aef6332b6904a","observation_id":"f256587f-fd18-4918-be5b-1e797c5c666e","resolution":{"observed_at":"2026-08-06T20:29:03.216236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.409054Z","title":"The effect of sampling temperature on problem solving in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.409054Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:4ce87ea3d257e0bae16a50ee6787167ac2a9157f921b57615121725d27185ba4","observation_id":"4eb03d92-b6a5-4658-90b2-9dd56e085279","resolution":{"observed_at":"2026-08-06T20:29:03.409054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.530733Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.530733Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:b542cdb8ef763d8788a05418c90670b695635d5a9e8e7e0573a9f17742d6cf7f","observation_id":"d70d4393-6219-408a-8892-8d96da99a79c","resolution":{"observed_at":"2026-08-06T20:29:03.530733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.675873Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.675873Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:998591081411da8a8ef5d48353593fbb58a796738adf4ee5f14c33f9663e5315","observation_id":"a4614c9e-01f7-4332-adb6-eec84c16fca5","resolution":{"observed_at":"2026-08-06T20:29:03.675873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:29:03.805548Z","title":"after incorrect reasoning or answer to prompt LLMs to self-correct, without finetuning. We observe significant reductions in the blind spot after appending ``Wait","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models","version":3},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T20:29:03.805548Z"},"links":{"citing_paper":"/paper/2507.02778"},"observation_digest":"sha256:947ec1df66288c40fa42a837119fa0e4291304823f42096ecf856e86c2279df1","observation_id":"e8dda0b1-32fe-40b6-bebf-104c5fd66015","resolution":{"observed_at":"2026-08-06T20:29:03.805548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.02778","last_updated":"2026-08-02T21:08:33Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T23:23:53.423769Z","submitted_at":"2025-07-03T16:41:30Z","title":"Self-Correction Bench: Uncovering and Addressing the Self-Correction Blind Spot in Large Language Models"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":0,"verified_fuzzy":15},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 4 inbound Pith citation observations for arXiv:2507.02778."}