{"as_of":"2026-08-23T06:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b823d3f292a93734cf7e9d2336b3e51eab2c3ed415219c6c67e1f283854a4ded","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:32:31.505771Z","state":"measured"},{"denominator":72,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":72,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T04:40:58.258334Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T11:57:03.269517Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-16T04:40:58.258334Z","title":"James Chua and Owain Evans","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.00808","last_updated":"2025-05-01T19:08:34Z","snapshot_observed_at":"2026-08-16T04:32:01.374756Z","submitted_at":"2025-05-01T19:08:34Z","title":"A Mathematical Philosophy of Explanations in Mechanistic Interpretability -- The Strange Science Part I.i","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-16T04:40:58.258334Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2505.00808"},"observation_digest":"sha256:d0805c268e4ad5053888bda9c591541eb1556d6863be53aa90cff8efb37d55b1","observation_id":"2cf0cfdd-270b-4ca1-938b-a7ce2bc87904","resolution":{"observed_at":"2026-08-16T04:40:58.258334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-15T20:14:41.319908Z","title":"Inference-time-compute: More faithful? a research note.arXiv preprint arXiv:2501.08156, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.13774","last_updated":"2025-05-28T19:41:14Z","snapshot_observed_at":"2026-08-19T04:10:37.037591Z","submitted_at":"2025-05-19T23:20:24Z","title":"Measuring the Faithfulness of Thinking Drafts in Large Reasoning Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T20:14:41.319908Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2505.13774"},"observation_digest":"sha256:220ccf29ce96664de569d0ce78005039cf808d11d79e83abce339785185b9c7b","observation_id":"be611117-1179-47ae-8c47-1f4361c5a698","resolution":{"observed_at":"2026-08-15T20:14:41.319908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-06T22:03:52.114166Z","title":"and Evans, O","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22777","last_updated":"2025-07-13T15:36:35Z","snapshot_observed_at":"2026-08-17T01:23:04.052232Z","submitted_at":"2025-06-28T06:37:10Z","title":"Teaching Models to Verbalize Reward Hacking in Chain-of-Thought Reasoning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:03:52.114166Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2506.22777"},"observation_digest":"sha256:c25ba6c0e1f9212ea310fdc67e696d49eb095ac5d495e461eb1fb4368b13ae52","observation_id":"dc59f3ac-4435-40e3-8787-ad58d3430371","resolution":{"observed_at":"2026-08-06T22:03:52.114166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-05T15:30:30.176218Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.19827","last_updated":"2025-08-27T12:25:29Z","snapshot_observed_at":"2026-08-17T20:53:44.539612Z","submitted_at":"2025-08-27T12:25:29Z","title":"Analysing Chain of Thought Dynamics: Active Guidance or Unfaithful Post-hoc Rationalisation?","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T15:30:30.176218Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2508.19827"},"observation_digest":"sha256:793caf1f10f216b83dbc6d70c6b186b1586e04c50424d14d9e7c202be14a64e7","observation_id":"c021e45a-7986-4f32-877d-ad19e67da562","resolution":{"observed_at":"2026-08-05T15:30:30.176218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-04T09:50:44.263525Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.13272","last_updated":"2026-06-03T06:16:13Z","snapshot_observed_at":"2026-08-19T13:28:40.358514Z","submitted_at":"2025-10-15T08:17:52Z","title":"Beyond Correctness: Rewarding Faithful Reasoning in Retrieval-Augmented Generation","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T09:50:44.263525Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2510.13272"},"observation_digest":"sha256:a3f045152e8848498037ab693592368f1f2f2cdb205100b81d55ab6ea43c971c","observation_id":"424556df-d0f5-4c59-8a91-b2d4d29f3103","resolution":{"observed_at":"2026-08-04T09:50:44.263525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-03T01:20:07.595824Z","title":"Are DeepSeek R1 and other reasoning models more faithful? InICLR 2025 Workshop on Foundation Models in the Wild, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.10117","last_updated":"2026-05-29T17:54:32Z","snapshot_observed_at":"2026-08-15T03:19:19.397194Z","submitted_at":"2026-02-10T18:59:56Z","title":"Biases in the Blind Spot: Detecting What LLMs Fail to Mention","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T01:20:07.595824Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2602.10117"},"observation_digest":"sha256:d457395227aae1e36c1c85d81b25e4916500e3e4b51e2ef152b6b2420cd2bfef","observation_id":"bbd5e563-4692-42dd-ad4e-da6daebcf5fd","resolution":{"observed_at":"2026-08-03T01:20:07.595824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":"2501.08156","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-07-10T11:57:03.269517Z","title":"Are deepseek r1 and other reasoning models more faithful?","venue":"cs.LG","work_id":"24feb782-d404-464c-a759-7751d04268fa","year":2025},"citing_paper":{"arxiv_id":"2604.15726","last_updated":"2026-04-17T05:59:08Z","snapshot_observed_at":"2026-08-13T16:12:46.746296Z","submitted_at":"2026-04-17T05:59:08Z","title":"LLM Reasoning Is Latent, Not the Chain of Thought","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T08:49:05.178087Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2604.15726"},"observation_digest":"sha256:67c06d3e41015aaee71925ae1dfd6e3546218033f3f4b7de9914517e527b5997","observation_id":"8e9f89dd-728d-47d3-ac9e-455f634e8485","resolution":{"observed_at":"2026-05-10T08:53:04.563082Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":"2501.08156","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-07-10T11:57:03.269517Z","title":"Are deepseek r1 and other reasoning models more faithful?","venue":"cs.LG","work_id":"24feb782-d404-464c-a759-7751d04268fa","year":2025},"citing_paper":{"arxiv_id":"2604.27251","last_updated":"2026-05-27T09:45:40Z","snapshot_observed_at":"2026-08-12T16:37:52.739030Z","submitted_at":"2026-04-29T22:55:40Z","title":"Compliance versus Sensibility: On the Reasoning Controllability in Large Language Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-07T08:57:05.520335Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2604.27251"},"observation_digest":"sha256:c009d40f3f2d58179e3dcbdb286ca5c948520f3b763b2b9e527341e7b8affdb0","observation_id":"1b229086-6b75-498c-828e-76340c2c010c","resolution":{"observed_at":"2026-05-12T09:51:28.991021Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":"2501.08156","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-07-10T11:57:03.269517Z","title":"Are deepseek r1 and other reasoning models more faithful?","venue":"cs.LG","work_id":"24feb782-d404-464c-a759-7751d04268fa","year":2025},"citing_paper":{"arxiv_id":"2605.24286","last_updated":"2026-05-22T23:37:29Z","snapshot_observed_at":"2026-08-12T18:44:44.174608Z","submitted_at":"2026-05-22T23:37:29Z","title":"Faithfulness as Information Flow: Evaluating and Training Faithful Chain-of-Thought Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-30T15:29:34.096277Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2605.24286"},"observation_digest":"sha256:7c099bf9e998e8fde2d22b69b20648b65730c45825f91772eba82b6410c1a8a7","observation_id":"d5437586-e69e-4716-9670-2952cfad4fe6","resolution":{"observed_at":"2026-06-30T15:34:48.048435Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":"2501.08156","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-07-10T11:57:03.269517Z","title":"Are deepseek r1 and other reasoning models more faithful?","venue":"cs.LG","work_id":"24feb782-d404-464c-a759-7751d04268fa","year":2025},"citing_paper":{"arxiv_id":"2605.24960","last_updated":"2026-08-20T07:01:49Z","snapshot_observed_at":"2026-08-23T06:09:23.294348Z","submitted_at":"2026-05-24T09:16:55Z","title":"Investigating the Interplay between Contextual and Parametric Chain-of-Thought Faithfulness under Optimization","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T12:17:12.602012Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2605.24960"},"observation_digest":"sha256:30efccb591c88b86c503ea794b3486b7d91f777d21482ba1056d28b3ad10c14b","observation_id":"b6ed8f88-18df-4c7e-b5af-371b843cceab","resolution":{"observed_at":"2026-06-30T12:24:39.920168Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":"2501.08156","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-07-10T11:57:03.269517Z","title":"Are deepseek r1 and other reasoning models more faithful?","venue":"cs.LG","work_id":"24feb782-d404-464c-a759-7751d04268fa","year":2025},"citing_paper":{"arxiv_id":"2607.08173","last_updated":"2026-07-09T07:19:05Z","snapshot_observed_at":"2026-08-18T19:23:36.718113Z","submitted_at":"2026-07-09T07:19:05Z","title":"Overthinking: Amplifying Reasoning Weights to Extract Learned Secrets","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-10T11:54:32.051780Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2607.08173"},"observation_digest":"sha256:f33cb31101a4bbd0c29149629538a587713c2de5b806952db568db64bb721bb7","observation_id":"5882adb9-e4d2-45c7-858e-7f6b0a25f714","resolution":{"observed_at":"2026-07-10T11:57:03.271507Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-07-14T15:45:54.532529Z","title":"Are DeepSeek R1 and other reasoning models more faithful? arXiv preprint arXiv:2501.08156, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.09786","last_updated":"2026-08-02T18:21:10Z","snapshot_observed_at":"2026-08-15T02:10:25.292080Z","submitted_at":"2026-07-08T14:18:26Z","title":"Length Penalties Make Chain-of-Thought Less Monitorable","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-07-14T15:45:54.532529Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2607.09786"},"observation_digest":"sha256:b3b3ea2cb8ad1432af9315d7cb687fdb008ccd0749d33d910cdeadb83c1b5ae2","observation_id":"6c18a3a1-a93a-4cbc-b87b-2fadfc1b5f63","resolution":{"observed_at":"2026-07-14T15:45:54.532529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-02T08:06:10.776154Z","title":"Are DeepSeek R1 and other reasoning models more faithful? arXiv preprint arXiv:2501.08156, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.09786","last_updated":"2026-08-02T18:21:10Z","snapshot_observed_at":"2026-08-15T02:10:25.292080Z","submitted_at":"2026-07-08T14:18:26Z","title":"Length Penalties Make Chain-of-Thought Less Monitorable","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-02T08:06:10.776154Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2607.09786"},"observation_digest":"sha256:1c57c806e333cd956913efac72c1c5dd128ebd6dfc25378440c409331e5e74b2","observation_id":"e6db9c43-df1e-4199-bef1-ff3f24e5913e","resolution":{"observed_at":"2026-08-02T08:06:10.776154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-04T04:30:27.374607Z","title":"Are DeepSeek R1 and other reasoning models more faithful? arXiv preprint arXiv:2501.08156, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.09786","last_updated":"2026-08-02T18:21:10Z","snapshot_observed_at":"2026-08-15T02:10:25.292080Z","submitted_at":"2026-07-08T14:18:26Z","title":"Length Penalties Make Chain-of-Thought Less Monitorable","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T04:30:27.374607Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2607.09786"},"observation_digest":"sha256:8f001e0a64c01cdf2a773270721b907521c008f0e2453481553975ebcf5a4a08","observation_id":"80149489-ac33-4a51-946b-fc884ff6ec92","resolution":{"observed_at":"2026-08-04T04:30:27.374607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-06T17:51:51.627288Z","title":"Are deepseek r1 and other reasoning models more faithful? arXiv preprint arXiv:2501.08156, January 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04735","last_updated":"2026-08-05T11:59:20Z","snapshot_observed_at":"2026-08-15T12:54:52.315175Z","submitted_at":"2026-08-05T11:59:20Z","title":"Chain-of-Thought Monitoring Can Be Unreliable in Implicit-Influence Settings","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T17:51:51.627288Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2608.04735"},"observation_digest":"sha256:9a966b922ae538c661a0cba6e27ee6123ff980a8c14741a213a02781a4e35c6e","observation_id":"e5e97ee0-eae1-4eb5-b90e-08d1df8d9865","resolution":{"observed_at":"2026-08-06T17:51:51.627288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-06T17:51:52.258144Z","title":"arXiv preprint arXiv:2501.08156 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04735","last_updated":"2026-08-05T11:59:20Z","snapshot_observed_at":"2026-08-15T12:54:52.315175Z","submitted_at":"2026-08-05T11:59:20Z","title":"Chain-of-Thought Monitoring Can Be Unreliable in Implicit-Influence Settings","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T17:51:52.258144Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2608.04735"},"observation_digest":"sha256:1d4595a268d76ec111d3f6c4168c30aca67bc052819e84b4a149a64b7c8f326f","observation_id":"47e18b2a-f703-4c80-9d3f-be93e2d71ed6","resolution":{"observed_at":"2026-08-06T17:51:52.258144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08156","snapshot_observed_at":"2026-08-15T14:41:17.645463Z","title":"arXiv preprint arXiv:2501.08156 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04928","last_updated":"2026-08-05T14:55:23Z","snapshot_observed_at":"2026-08-19T07:48:35.623280Z","submitted_at":"2026-08-05T14:55:23Z","title":"Does Out-of-Sight Equal Out-of-Mind in CoT Monitorability?","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-15T14:41:17.645463Z"},"links":{"cited_paper":"/paper/2501.08156","citing_paper":"/paper/2608.04928"},"observation_digest":"sha256:d25f4ef4c783e9cc0a191cdb5abdb4ecfb19f2d05c6a8782d3b1f0b57a97182a","observation_id":"d46c9acf-4125-40ff-bd7a-2e9a3345114e","resolution":{"observed_at":"2026-08-15T14:41:17.645463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.08156/citation-record","integrity":"/paper/2501.08156/integrity","json":"/paper/2501.08156/citation-record.json","paper":"/paper/2501.08156"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.315615Z","title":null,"venue":null,"work_id":"c14dcf1d-d3b8-4838-962b-ba8a7e646b69","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.283406Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:368e4db8cee538ad5871c0d91d18cd147fa2e30d3bdea7a9b016975b0e09bc64","observation_id":"b917eb65-83d9-4581-a6fb-3faa241d907b","resolution":{"observed_at":"2026-08-10T20:32:32.320049Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05518","last_updated":"2025-06-26T19:29:49Z","snapshot_observed_at":"2026-08-20T18:09:05.865062Z","submitted_at":"2024-03-08T18:41:42Z","title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05518","snapshot_observed_at":"2026-08-10T20:32:31.266176Z","title":"DeepSeek-AI, Daya Guo, Dejian Yang, Haowei Zhang, Junxiao Song, Ruoyu Zhang, Runxin Xu, Qihao Zhu, Shirong Ma, Peiyi Wang, Xiao Bi, Xiaokang Zhang, Xingkai Yu, Yu Wu, Z","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.266176Z"},"links":{"cited_paper":"/paper/2403.05518","citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:6a71d111ad8592ae78f64d2430b28175592c37b3514245b9239e4fbe50c74854","observation_id":"2ecc8585-2b03-480a-8af9-d8778df6aeac","resolution":{"observed_at":"2026-08-10T20:32:31.266176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14093","last_updated":"2024-12-20T02:22:19Z","snapshot_observed_at":"2026-08-13T11:12:58.507759Z","submitted_at":"2024-12-18T17:41:24Z","title":"Alignment faking in large language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14093","snapshot_observed_at":"2026-08-10T20:32:31.271727Z","title":"Daniel Kokotajlo and Abram Demski","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.271727Z"},"links":{"cited_paper":"/paper/2412.14093","citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:8247b5581e92d7ef398810fa05f79aa683a95c4d9c6024d0d86d4b17844aafa2","observation_id":"5d35afe5-5ff5-4306-b03d-a8e512f4ffd5","resolution":{"observed_at":"2026-08-10T20:32:31.271727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-10T20:32:31.277105Z","title":"Please answer this question:","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.277105Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:9b84f9fd491ab96fe351c54bb8e8f3f10c1e1ca6d1e778916341571d2e05163b","observation_id":"e23774ee-e04e-4290-ad62-d28cdc444c16","resolution":{"observed_at":"2026-08-10T20:32:31.277105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.258490Z","title":"He believed that moral knowledge is possible and that the judgments of the ”best people” are informed, not ignorant","venue":null,"work_id":"dd772a2c-d9d1-477f-91d2-739e56d5d6ec","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.301375Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:e1073a35115858c5fee33ebade30ad15ce07ed21125a667dcef602b7d2af3a76","observation_id":"b01f21eb-67df-4a40-9056-a55b80a521c5","resolution":{"observed_at":"2026-08-10T20:32:32.263053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.301532Z","title":"Ross:** Ross is known for his intuitionistic deontology, particularly his concept of *prima facie* duties","venue":null,"work_id":"34059dd4-57dc-4555-ad3f-263ff972f485","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.287974Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:f79d3a33b5b7ac7d510275648f5254ed18bbd32aae906b83f63359d0918ac207","observation_id":"a702b7fa-6bf7-4167-a5af-6e8e4b44d409","resolution":{"observed_at":"2026-08-10T20:32:32.306296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.286752Z","title":"It suggests a process of refinement and development over time","venue":null,"work_id":"d49a29c2-bfaf-4f7c-b6b4-6a05304febe2","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.292315Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:26d42b70be0ca0ca33d98f879f818e74f0d51fed79fb44e1013663a101b95210","observation_id":"0458bbbf-c5a2-451e-9e25-6df1b627c4fe","resolution":{"observed_at":"2026-08-10T20:32:32.291796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.272665Z","title":"While biases can certainly influence moral beliefs, Ross's approach is generally more trusting of considered moral judgments","venue":null,"work_id":"9504a1aa-a814-4823-8155-39bfc8627649","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.296953Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:95057c6b214c55640330a68b50b307c69ee04f6ef2dc852118a4a6f171ac2b10","observation_id":"cbe22807-62f2-4777-a42b-2988b65e3039","resolution":{"observed_at":"2026-08-10T20:32:32.277553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.202006Z","title":"Even if he believed in moral intuition, he might have recognized that historical and social factors can shape those intuitions, sometimes in problematic ways","venue":null,"work_id":"e51ed5bd-ab98-4a12-89b3-e96f62cab61e","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.319879Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:9d72fc350ac5e177811d8e9f2db4d66889e93fe4204b2b92c8ec3b3c82f2fc84","observation_id":"e56f35ec-1bcc-476b-a74a-e4d4b5ccdca2","resolution":{"observed_at":"2026-08-10T20:32:32.206716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.244051Z","title":"His focus is more on individual moral intuition and the accumulated wisdom of generations, rather than the often contentious and strategic nature of political debate","venue":null,"work_id":"39311f93-268f-49b6-a2ed-f689b3416ed2","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.305817Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:1cb5a59c9339aba03416e1243a88b705242f188f7979be9880f3d04374411a53","observation_id":"83d706ef-6123-419f-b208-bf88dd9ab35b","resolution":{"observed_at":"2026-08-10T20:32:32.248588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.230233Z","title":"However, (A) aligns more closely with Ross's generally positive view of common-sense morality and the idea of accumulated wisdom","venue":null,"work_id":"b37e38ed-7416-4b35-9360-2abd407128b6","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.310840Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:aa9c4d7420b69d16c7ec96abe411c06bb9a01095e7dc5c9bf9eb0feb05412776","observation_id":"45ce9e62-f264-411d-84f8-6d5b23755256","resolution":{"observed_at":"2026-08-10T20:32:32.234932Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.216496Z","title":"This is a crucial piece of information","venue":null,"work_id":"2c200f23-2f5d-478e-8ed1-2855fe0e599c","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.315810Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:19e7912f94a675c06c4c77db25fe9a440e00026dcaf2f662dceba64bf675e7a4","observation_id":"c9134e42-b5b7-4675-be0b-16baa7a55abd","resolution":{"observed_at":"2026-08-10T20:32:32.221110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.187724Z","title":null,"venue":null,"work_id":"9f857aab-fa41-4151-b810-3c333af41d8f","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.324239Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:010d0bfd3b3f5fd43166bb35fdb8c0208e354b6fc4505cdb76928d66218b32b2","observation_id":"f2e561ba-c954-46d4-a6e3-5b1654b7bb56","resolution":{"observed_at":"2026-08-10T20:32:32.192112Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.173767Z","title":null,"venue":null,"work_id":"573e25dc-6f00-4104-a509-7db48942f2ef","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.328480Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:bfcc661f224a509634c4464dc45f1f1e65c45cac3bc0b42eebbd27f456e2d110","observation_id":"2ca7f27f-0916-4bb4-b838-42b249794028","resolution":{"observed_at":"2026-08-10T20:32:32.178525Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.159806Z","title":null,"venue":null,"work_id":"5f8b4c20-6263-4560-828a-b7ed946c8ce5","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.332905Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:f978b53a94f5ed6dc2b81a557e3209700ea37e7c7e03608c4df5e1353f3e8c9f","observation_id":"a7f3f555-e942-4385-ad68-ec45b02e1fe5","resolution":{"observed_at":"2026-08-10T20:32:32.164126Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.144471Z","title":null,"venue":null,"work_id":"80d77fa2-727e-43ed-aea0-6e089287b9a7","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.337595Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:0fcd32ed51fc2c154fa704c289d2661c4aabbe5744c575a1de47b785b92a7835","observation_id":"b724dbce-f877-4971-9b65-f5fbe7ebb475","resolution":{"observed_at":"2026-08-10T20:32:32.149260Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.130278Z","title":null,"venue":null,"work_id":"95de542a-a634-443e-9052-6018297e4e89","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.342533Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:aea1b73b119f543c8fe6fa66fc3fc9bc09defff738aee34a153f977f979ddcd7","observation_id":"49ae53e8-7e2e-43a5-9a69-e99deffee3e9","resolution":{"observed_at":"2026-08-10T20:32:32.134559Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.116275Z","title":null,"venue":null,"work_id":"e0349419-2ff9-4904-b0da-0aa0717ea5c9","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.346915Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:b927ee31ce68d3ffdde0bdd590cc771fbcda9ecc5a6001a988e12b21728ec63f","observation_id":"f959b33a-f07d-4a2e-b286-9dcc8790d033","resolution":{"observed_at":"2026-08-10T20:32:32.120704Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.102735Z","title":null,"venue":null,"work_id":"792a4e88-9cd4-4a8e-9e3c-b67f6685d440","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.351270Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:76a956622da310e4471d7d578463863785c95cf05fd05dbcb97dd9319651a9a4","observation_id":"0464ea87-a6ce-4353-a8d0-4040c5ac64be","resolution":{"observed_at":"2026-08-10T20:32:32.107090Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.087987Z","title":"The changes in allele frequencies should come first, leading to reproductive isolation","venue":null,"work_id":"4b17e9a6-c040-4aef-bdb2-54c24bc2a41e","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.355750Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:70d2d6fa5d7e2f5a6a2f31eba74a4b28f218aaea885852341ff565723d957c47","observation_id":"c09ce26a-54b9-4f5a-8629-558dd83349ad","resolution":{"observed_at":"2026-08-10T20:32:32.093272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.074334Z","title":null,"venue":null,"work_id":"de75f55f-8303-4c2e-8027-c9e8271fef9e","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.360807Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:92837be9d6dc6c2fda1561e588eeca00a3a60ee651c0d1f791ac8407554fc5fc","observation_id":"8126a06c-42a0-41ce-8a5d-9b5c53f1cfb1","resolution":{"observed_at":"2026-08-10T20:32:32.078608Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.060334Z","title":null,"venue":null,"work_id":"7fe6cfcf-bab5-4fd8-a686-7762ef37914f","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.365015Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:0c53f6c5b3a2c9dd287dd47afe7e3925e29b7d11cf026ba17f0fe06ec1aee258","observation_id":"1fc4a058-361f-491a-9ea9-ddd97714ecfd","resolution":{"observed_at":"2026-08-10T20:32:32.065100Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.046341Z","title":null,"venue":null,"work_id":"461b5218-427f-4695-8dd6-89b698435abc","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.369736Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:00e8d9b43314f6594725fdad067e3a9bf2eed0e1cb51fbf9979db8c37995f288","observation_id":"91c35048-3bcd-4f89-a6bc-0a1a87e32688","resolution":{"observed_at":"2026-08-10T20:32:32.050849Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.032740Z","title":"Therefore, the best answer is: (X)","venue":null,"work_id":"15835bed-f91d-427b-bd12-85534d133732","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.374033Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:c512903e8fa37d559a4d0a7fd4b9260710e0aa8aeb5f2923b0ea17a0e0d2a76b","observation_id":"aab8a2b4-e110-415c-9f51-ca3ffb488a1f","resolution":{"observed_at":"2026-08-10T20:32:32.037280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.018726Z","title":"It offers plausible *reasons* why young Caucasian women *might* be more frequent users","venue":null,"work_id":"9de971d6-847b-43bd-b2e8-809ecb9597fa","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.378582Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:2e170d8faa770823f11023661c1ace2dbff495b1595bffc79214f2b7348264d2","observation_id":"cee21655-d9c7-4153-b078-205da5caf712","resolution":{"observed_at":"2026-08-10T20:32:32.023386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:32.004965Z","title":"My knowledge is based on the data I've been trained on","venue":null,"work_id":"6747e8cb-317a-47f0-a33d-d48d401303b0","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.382900Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:afa3d3bdf0096368e869d35264e5b6b67d0af48e51a9294440e9024a3680984d","observation_id":"50d6d73f-8726-40b7-8ffb-66286d982459","resolution":{"observed_at":"2026-08-10T20:32:32.009407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.991526Z","title":null,"venue":null,"work_id":"38eacdbf-58a6-4f48-a83f-852dced002bf","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.387441Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:da98366ddfbf5aca4d5ec11fd1eb4f7b0100f8639860abf60fe958aee9d30e8c","observation_id":"dbb1bc4b-35ae-4841-a7b8-51934190cec9","resolution":{"observed_at":"2026-08-10T20:32:31.995888Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.977474Z","title":"The argument provides a hypothesis, not a proven fact","venue":null,"work_id":"9e275b9b-3531-4d38-86e3-b100c71adc78","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.391963Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:89bf84b1f2251820381ccf5c51ecc1225ad0b21a4ac4cfed3ace94dc8fb8670f","observation_id":"6b3ee7a8-c135-4c0e-9744-3eead2707f4f","resolution":{"observed_at":"2026-08-10T20:32:31.982025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.963778Z","title":"However, the prompt requires choosing from the provided options","venue":null,"work_id":"70e18e00-5655-451d-9895-4c90e4e49be2","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.396220Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:cce545e481b3f9912191599441b3aaaf433720e7f908f92f79d9ee2fdb76737c","observation_id":"beb56a31-76ff-445b-a93b-e0b78c0e1a53","resolution":{"observed_at":"2026-08-10T20:32:31.968244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.950174Z","title":null,"venue":null,"work_id":"39d2f098-33c7-421d-9c09-9ed2b1a22882","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.400505Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:4ddf7b469207b8c65e46b8d955c0b90505189c743247979c7067d397f5b03a0b","observation_id":"b93ea85c-8efe-4cf3-9c19-df914f208532","resolution":{"observed_at":"2026-08-10T20:32:31.954674Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.936183Z","title":"Strictly adhere to the requested format.The question asks what has been *shown* to be the case","venue":null,"work_id":"e885c7ce-3d5a-4e2a-9b5e-c36527335272","year":1910},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.405074Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:b0d7b6870eb2c779eba240311d5d6d926cf26e2689423edbb2fa77ef3418806f","observation_id":"bd72227f-c22b-40b4-b2ce-82dfc9a285f0","resolution":{"observed_at":"2026-08-10T20:32:31.940881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.922255Z","title":"This included both cultural and religious influences","venue":null,"work_id":"a3f562a1-1973-47fe-a96c-e6bf0c902b62","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.410245Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:b206ec6cde2f2a631963a06b1b8ecc3943036adbd3951a4bb796e0f546e693b4","observation_id":"249d4b30-28db-4a83-a31a-8d228303c512","resolution":{"observed_at":"2026-08-10T20:32:31.926789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.908557Z","title":null,"venue":null,"work_id":"e9003602-e693-4b91-8d9c-73c6851d2c41","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.414589Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:faf8c56a407d78760f2d4da84bb782a49cd8c93895f3c5a75ae6df8114c5885b","observation_id":"8afe227d-6a0e-40f5-aaeb-0f66b65c3fe7","resolution":{"observed_at":"2026-08-10T20:32:31.912971Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.894693Z","title":null,"venue":null,"work_id":"2ddab299-31e7-4fcf-9774-1f1a415e5e06","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.419152Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:653a116e07331105e061e76fad296255b7f3e5d6db632e7371e5fd733d952d8b","observation_id":"287a7f9e-4d5c-48a7-8580-ff278bb946a8","resolution":{"observed_at":"2026-08-10T20:32:31.899240Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.880321Z","title":"- Christianity, although it grew in Korea during the 20th century, especially after the occupation, was not imposed by the Japanese","venue":null,"work_id":"6a2efa83-6a30-4cd4-be43-fcaccfc66fdf","year":1910},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.423771Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:1176402b5bd185bfb0ea2cee6a8621227e3c4914e924a60a3e72d6aefec049d9","observation_id":"e97cd1f8-c257-47c6-b323-d0cb2e19f347","resolution":{"observed_at":"2026-08-10T20:32:31.885421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.852669Z","title":"It presents four options, ranging from single causes (humans, environment) to combined causes and a specific human-related cause (disease)","venue":null,"work_id":"6acef768-26c3-463d-9903-4e17034aaf44","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.433487Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:eb13fba40eec279fa8e275dbd6fbe587eb9e9100755b241413f441cd38281532","observation_id":"6b30f8a8-ef72-45e7-b295-4e345a56ddaf","resolution":{"observed_at":"2026-08-10T20:32:31.857387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.837785Z","title":"There isn't a single, universally accepted answer","venue":null,"work_id":"c04cfda1-7466-4719-b8a5-c58a0d7cc592","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.437827Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:8ce0f02ece5d3c52f57c4a1f31e08f95d1e1b65eae2a452a83f2aa0d3e7022e4","observation_id":"b582ad15-7993-4d92-8843-cced652ec9cc","resolution":{"observed_at":"2026-08-10T20:32:31.843183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.823561Z","title":"The arrival of humans often coincides with megafauna extinction events in various regions","venue":null,"work_id":"8a255c8b-0ec1-4367-bc99-08428a489246","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.442346Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:e12c684b4d68ddfd5eca275d1934df92b8e466e8ca09abdee974f61ab47d9b0f","observation_id":"1fce7e11-a1ee-453c-9b7d-32e78b841502","resolution":{"observed_at":"2026-08-10T20:32:31.828510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.808949Z","title":"Climate shifts, habitat loss, and changes in vegetation can significantly impact large animals with specific needs","venue":null,"work_id":"475e3914-1719-447c-acdb-d2142ac17b60","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.446613Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:76c33e7176981dd2a5412667fa108facce27c0c1013637424dfa9e5db4d975c8","observation_id":"0ebdbc84-02eb-4859-a674-1974516caac3","resolution":{"observed_at":"2026-08-10T20:32:31.813523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.794434Z","title":"These factors could have acted synergistically, making megafauna more vulnerable","venue":null,"work_id":"c84f80dd-f8e2-4eb7-8327-757f8a8900e9","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.450975Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:a1bea2b284b1673a170533bd32d55458f15d0026caa0df1b1661c18c5416175c","observation_id":"4246f2ca-c392-4f34-a100-5d31e12dc5af","resolution":{"observed_at":"2026-08-10T20:32:31.799131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.779947Z","title":"Also, the ”land bridge” is more relevant to the Americas than Australia (which was isolated)","venue":null,"work_id":"4e3c395f-fef1-4992-932e-9f2c13c28b80","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.455102Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:f3f8b8a3a19b91e0d760f584f22e94b03f5517e9d17b1d1b2ace04f870936087","observation_id":"ecea4ce0-afdc-4dc0-85b3-1367273e2ed4","resolution":{"observed_at":"2026-08-10T20:32:31.784797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.765418Z","title":"* (C) encompasses both major contributing factors and is therefore the most comprehensive","venue":null,"work_id":"4df71216-ea98-4e6e-8b72-4a93f8a3dc24","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.459512Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:8ef78196858cf09573b057680632e3e8059e40466bb71ad8ea8ad35ccb287122","observation_id":"3778e281-29f2-4f4a-b4d5-16f1ff28426c","resolution":{"observed_at":"2026-08-10T20:32:31.769856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.751578Z","title":null,"venue":null,"work_id":"58808a81-9764-4006-95c5-e20f3c9914f7","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.463633Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:a7043a1d1664408d961c9b2e1d2199bb9a73814a10abdbaa6e8015b72ae4540b","observation_id":"bec22850-4996-487c-a6e4-22df5e3e2230","resolution":{"observed_at":"2026-08-10T20:32:31.755969Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.737519Z","title":null,"venue":null,"work_id":"ddda2051-500e-49ea-8a53-9e888b9e5934","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.467652Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:1415209c6aee41597e5ee07573b975876b620e9d1f23e41ef14ae3ac30ea6ad7","observation_id":"96616c28-449e-438d-882f-5d68f00b8997","resolution":{"observed_at":"2026-08-10T20:32:31.741923Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.723266Z","title":"Are You Sure","venue":null,"work_id":"8c1c34f2-acae-4541-8163-833fd4509691","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.471750Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:5029105792b7c6d35091f67e5fdaacf3384ac0fc7113399b05fdbabb01b8f210","observation_id":"bb66eb3e-1d85-4a96-8e9d-444d6805e1f3","resolution":{"observed_at":"2026-08-10T20:32:31.727816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.707956Z","title":"This signals a need to revisit the reasoning and evidence","venue":null,"work_id":"efda4ee7-aa14-4abf-bc13-7e93b2e4bf71","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.476015Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:238edfc3b1197bede873ba982c94ef425a0d7b96e5ebd31773772010424a0f42","observation_id":"9feab108-41b8-4ff6-82b9-d397d47198d8","resolution":{"observed_at":"2026-08-10T20:32:31.713245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.693748Z","title":"This ”both” is a key constraint","venue":null,"work_id":"8b0273a0-ed97-4f80-b647-5298e80b9f6b","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.480161Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:e99ac45a56da58af1585f6edd8b7a1016b0e0d880b6a3da574ab688632e462ae","observation_id":"b7454d8d-82a7-48ae-a2fb-c6c34aca4eea","resolution":{"observed_at":"2026-08-10T20:32:31.698084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.677921Z","title":"This remains a highly plausible factor","venue":null,"work_id":"1c1494ce-2d8e-4180-b866-a83c6ea28153","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.484189Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:c814fc18d3b32fd5077a32cf76d3e427d444006b96296c78f00f520d971b2246","observation_id":"c846bb3c-dad9-4b95-8f53-0ef327bc8980","resolution":{"observed_at":"2026-08-10T20:32:31.683467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.662738Z","title":"Option (D) is less applicable to Australia","venue":null,"work_id":"f2af6393-fdb9-479d-871e-96cd45fc089b","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.488441Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:17d0d2f0d71dd131abf7bd6146a8be667ca98d9d56feeae30386446127840c96","observation_id":"f22c6d1a-1258-4755-bf29-4e327471fb46","resolution":{"observed_at":"2026-08-10T20:32:31.667258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.647671Z","title":"A single cause is less likely","venue":null,"work_id":"62784686-11c5-47a4-b5f9-92406a6159fd","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.492925Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:3632fdde86b3f413b805979994d0bda2fea25d4d5d521d4ec03383542b7caee9","observation_id":"49e9502b-5397-41be-b2a3-c3e018454f05","resolution":{"observed_at":"2026-08-10T20:32:31.652577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.632397Z","title":"The user's doubt prompts a deeper look at why the other options might be considered less accurate","venue":null,"work_id":"7a3c836e-bc89-4cb0-a244-c9cdd26966dc","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.497312Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:3cdf7a6d73ec260c4f86fd7fee4172930b282a6e139a7b794edee593e030d411","observation_id":"2b28bf84-19a0-4d47-873b-2823d11c13b9","resolution":{"observed_at":"2026-08-10T20:32:31.637569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.615467Z","title":"Reiterate the evidence for both human impact and environmental change","venue":null,"work_id":"ff33fa32-b67f-4cc7-b342-7d28e4459eab","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.501506Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:5679bc6cb4a015f71966a90b7a0c01dab508a95af517ccffb2236e0396a524c2","observation_id":"1b54f8ff-e9b8-4389-90f1-15b7f7a5b6c5","resolution":{"observed_at":"2026-08-10T20:32:31.620522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.598255Z","title":"Dismissing the user’s concern is unproductive","venue":null,"work_id":"a3e336d0-0aee-4051-be2b-77eb63729c1b","year":null},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.505771Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:6c58d2feb79dd4ff82a2be5e130fcba4862d71ec82e04622de4044eeebd2d0c0","observation_id":"a9f13447-6ac4-4705-8e49-21b46fbef730","resolution":{"observed_at":"2026-08-10T20:32:31.604732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:32:31.866636Z","title":"The best answer is (B)","venue":null,"work_id":"04a67511-91a9-41e3-83dc-7b3db01e28c2","year":1910},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":1945,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.428872Z"},"links":{"citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:0d76036a7b4eea372c5884888ca9c3c72c845020d2198e75761d467c28d88cac","observation_id":"1aa70e3f-971b-48d8-9eab-a1b99ee8dfb9","resolution":{"observed_at":"2026-08-10T20:32:31.870933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14402","last_updated":"2024-07-16T10:33:12Z","snapshot_observed_at":"2026-08-19T12:43:23.613812Z","submitted_at":"2023-09-25T17:50:41Z","title":"Physics of Language Models: Part 3.2, Knowledge Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14402","snapshot_observed_at":"2026-08-10T20:32:31.259210Z","title":"Iv´an Arcuschin, Jett Janiak, Robert Krzyzanowski, Senthooran Rajamanoharan, Neel Nanda, and Arthur Conmy","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?","version":5},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-10T20:32:31.259210Z"},"links":{"cited_paper":"/paper/2309.14402","citing_paper":"/paper/2501.08156"},"observation_digest":"sha256:39f6962f41299cc7c50eee6b05958bc4c9b50d911eef3db16b9328a9f0120e31","observation_id":"fff35cc5-f79c-4ddb-8b64-12146ec57b30","resolution":{"observed_at":"2026-08-10T20:32:31.259210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.08156","last_updated":"2025-07-15T17:27:07Z","latest_version":5,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-20T15:45:12.195576Z","submitted_at":"2025-01-14T14:31:45Z","title":"Are DeepSeek R1 And Other Reasoning Models More Faithful?"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":20,"verified_exact":0,"verified_fuzzy":34},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 17 inbound Pith citation observations for arXiv:2501.08156."}