{"as_of":"2026-08-05T21:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b24bf3324b4aeefe918fef0981b16f84293ead6b6f4935ff0bf852ee971ec474","coverage":[{"denominator":108,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T03:56:28.073779Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.13753/citation-record","integrity":"/paper/2607.13753/integrity","json":"/paper/2607.13753/citation-record.json","paper":"/paper/2607.13753"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:16.720764Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:16.720764Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:525e3de830d5afb5058f6d00ab9cf164afba56e06ea7819934bbdf87ff8a2596","observation_id":"443ca3e7-4616-46c7-96cb-be21c0a9fc7c","resolution":{"observed_at":"2026-08-02T03:56:16.720764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:16.883058Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:16.883058Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:18385b45d7d6bbfa7ffb075374a0a646155c2777ab2f841f54fde949ee30fff8","observation_id":"8137768e-02ec-4d3a-afb5-67e47d2f671f","resolution":{"observed_at":"2026-08-02T03:56:16.883058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.046894Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.046894Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:cd4a560ba6af5564d81db24451b65da6724ce87cf6c6e93872d46596d5bec6a7","observation_id":"b818d33e-8ba0-4a99-ba8b-dd3b1a83e111","resolution":{"observed_at":"2026-08-02T03:56:17.046894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.305342Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.305342Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:a64613bcf627b94f91c412cc4024b0566f07f92868a640ad5255c021194161d5","observation_id":"995417c0-b4b1-426d-a748-ff81dcc1857a","resolution":{"observed_at":"2026-08-02T03:56:17.305342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.525161Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.525161Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c70827b552fd1f4cf03d50a4b7ff98010faf4a054a80cdad1e8a19314d9b7f43","observation_id":"4ee00f7c-e497-4795-8696-7bf719da8121","resolution":{"observed_at":"2026-08-02T03:56:17.525161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.635578Z","title":"2021 , eprint=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.635578Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:332d4972441772701cd5416a2731dc1c9405d4a9511c2d7e302c0c130506ebc4","observation_id":"303f3ccf-1834-4265-9893-814d146fa234","resolution":{"observed_at":"2026-08-02T03:56:17.635578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.746809Z","title":"2022 , eprint=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.746809Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:f653e88b2631a0c703771b000deec2213989c49136e61c8037591cd93c0bd96f","observation_id":"9bbd247a-6fae-4182-af5a-2a9e713390b5","resolution":{"observed_at":"2026-08-02T03:56:17.746809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.858651Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.858651Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1aeabf25ad870693e891d53a79e03a7799a8fbf04cd7e266333ed6cc538b5609","observation_id":"481f8c39-189d-4581-8f4e-39db76134d38","resolution":{"observed_at":"2026-08-02T03:56:17.858651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:17.970359Z","title":"Findings of the Association for Computational Linguistics: ACL 2025 , pages=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:17.970359Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c6136efdf011b135e37eda422728d73e74a0c7f99a7a6feba8ead89b44aa00a7","observation_id":"a5aedc63-fff2-49b5-82b3-d7e73f61193d","resolution":{"observed_at":"2026-08-02T03:56:17.970359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.081629Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.081629Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:94bc1479f98bcf095ab49853518026ba4914b923e47c161f10949b6fc9f5751f","observation_id":"8db46a12-1db6-41bd-9b33-2135fa964d2b","resolution":{"observed_at":"2026-08-02T03:56:18.081629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.155745Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.155745Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2664929fefa9a0fe809de68b2c5411df2aa7aaa0b7fdfb29ea8d2e8221dad558","observation_id":"14a5ed35-9b7e-417b-9ebc-e76c0a4a07e8","resolution":{"observed_at":"2026-08-02T03:56:18.155745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.269368Z","title":"International conference on machine learning , pages=","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.269368Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:329027080b6cfdb8fb4ece656f28a9fc34c6d6880b888c2e8c8d744947c56008","observation_id":"d5eaecc8-914e-4465-a95f-2b7cfeb643e1","resolution":{"observed_at":"2026-08-02T03:56:18.269368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.344443Z","title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) , pages=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.344443Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:dd1cd11bc9533e5c1948aad3e3c9907a9dc3430c7bb1383a30d673c99d04ef30","observation_id":"70570ddf-5a85-4d37-877f-27f3bf62b6ed","resolution":{"observed_at":"2026-08-02T03:56:18.344443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.454309Z","title":"Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.454309Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:319adcf49fc8bcdda66f203fbf67659339c8e5e39023fd2990541d6fb1521cbd","observation_id":"6db761c1-bac1-4316-9bb1-20e4ce62ceb3","resolution":{"observed_at":"2026-08-02T03:56:18.454309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.671429Z","title":"2022 , eprint=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.671429Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:48ffb133d1006068af26e76a4c85271c1497099a93232b4299ecc0f5ab17d6ba","observation_id":"ae9d9765-a8e3-4738-9e46-66826995793e","resolution":{"observed_at":"2026-08-02T03:56:18.671429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:18.942951Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:18.942951Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:e5c0c31a1dbc6cc29706f8821169ebb7c86df595d191058f48906e6c27858552","observation_id":"1553f37e-77f9-475d-a7cd-62e039ee26cc","resolution":{"observed_at":"2026-08-02T03:56:18.942951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.030784Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.030784Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6e24b118f53c72fe273a83016dbf7db059f16379688117333a5ff047986fa95c","observation_id":"35799c2f-73dd-4fcd-957d-f4d650b189a9","resolution":{"observed_at":"2026-08-02T03:56:19.030784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.161410Z","title":"and Hauskrecht, Milos , title =","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.161410Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ef45b2440bd84ff93b32a98211b9c8170e6a986799a9f560a302dfb987094dc2","observation_id":"0d752492-ea31-4e86-a8d5-2bb895aa4e13","resolution":{"observed_at":"2026-08-02T03:56:19.161410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.236437Z","title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP) , pages=","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.236437Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2b5c2fb6264c9bb6326642e7f877778d7db6ea0e3007eb6f31b4ce44c8792bfe","observation_id":"bfa3907a-f566-461b-bf16-2ecafcc4d116","resolution":{"observed_at":"2026-08-02T03:56:19.236437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.311265Z","title":"Transactions of the Association for Computational Linguistics , volume=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.311265Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ece5055662ccd14a43210721afd7e5c6846f8bfb2b9edef0e7ed89fc03349e13","observation_id":"2d482648-1523-4db9-949e-092c661cb44d","resolution":{"observed_at":"2026-08-02T03:56:19.311265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.357682Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.357682Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6adc7228bfaf121ba31308e6e3618b5c20d701f543e2f4736e8ea2d7c9c2d1bb","observation_id":"de624713-9517-46d4-8228-da746102f819","resolution":{"observed_at":"2026-08-02T03:56:19.357682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.405028Z","title":"Probabilistic Outputs for Support Vector Machines and Comparisons to Regularized Likelihood Methods , volume =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.405028Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:eaeeabcbf23e940497cf61178408d77a2a59de476a3dd9d51131770c6fc33749","observation_id":"5872479e-bfa7-4646-9170-0f548b7c7fc8","resolution":{"observed_at":"2026-08-02T03:56:19.405028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.467260Z","title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.467260Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:0ddb1033f015f40badd38522f89cedd605af7dc29ae8257626f9aa7583529651","observation_id":"2931a62b-0358-4aef-a985-a22bbc4e8700","resolution":{"observed_at":"2026-08-02T03:56:19.467260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.539479Z","title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.539479Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:516975d6fc5573bdd0ef4cc348cb93c49973bcecfc6560c85b818e296e0cf4b5","observation_id":"1d21ee5c-4e73-461c-9dd0-3a159e64e037","resolution":{"observed_at":"2026-08-02T03:56:19.539479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.688306Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.688306Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1125f8e7caf1e70acac8e91cd954ff88b3b5c1a49b13970840f5f8d001173933","observation_id":"14429ecd-e4fc-40fb-bdd2-dcf829f58a32","resolution":{"observed_at":"2026-08-02T03:56:19.688306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.785245Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.785245Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:5bc501e8323d528be24dd1873a015010f51a55bab476cf682251496eaa2c449d","observation_id":"b8358bdf-efb0-4455-9865-cf423c91d429","resolution":{"observed_at":"2026-08-02T03:56:19.785245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.841972Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.841972Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6f1481fe30827784964803d5a2e84d4309c35e3e03ab15728c79d051ea55719e","observation_id":"a8f3c383-7b93-48bf-8d1b-36fbef7d563e","resolution":{"observed_at":"2026-08-02T03:56:19.841972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.896437Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.896437Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:986947132866c75b7bae84b1a20f8cdd0161f09a897aafe41110c13108f504e2","observation_id":"3cfe850b-d513-4f74-ba7d-7fe46af8b78d","resolution":{"observed_at":"2026-08-02T03:56:19.896437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:19.967258Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:19.967258Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:f3a5c4f80f3e465ff04130253a85047c357bfbc766a01770289dc90128e47d44","observation_id":"2ec03989-89e6-4858-bd76-777660b7ef7a","resolution":{"observed_at":"2026-08-02T03:56:19.967258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.024839Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.024839Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:aec535b812a1d62bf9d12ab1ed50c767e18af4d40423a82052908f3c8d3c5dbd","observation_id":"23f17299-ffe3-49a9-b8d2-093aca93aa4a","resolution":{"observed_at":"2026-08-02T03:56:20.024839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-02T03:56:20.086170Z","title":"arXiv preprint arXiv:2501.12948 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.086170Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:0780289ed4dfd50f61aede0c36b02e69873c25fc596f045fcd5d776af759e805","observation_id":"d0caa55b-9345-4fce-b06c-cf68ef045cb4","resolution":{"observed_at":"2026-08-02T03:56:20.086170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.204278Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.204278Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c3aabbaddd432427a3b994b55106cc8f828d9d3e6be3913104a39da8af5eb3a8","observation_id":"bbb41a7b-cb88-49c8-b141-d2a90c8abb8e","resolution":{"observed_at":"2026-08-02T03:56:20.204278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.271780Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.271780Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:4ce9db9e6c29dbddffe52e0627071ba5ee6399e4ee2e9fc8ebec518b35649358","observation_id":"ba77f5a3-39ae-42c2-9381-757c04c03ad0","resolution":{"observed_at":"2026-08-02T03:56:20.271780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-02T03:56:20.714439Z","title":"5-coder technical report , author=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.714439Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1e3230801e0bf2ad2361dc4d89763d4a64b2bf1896d542b2ef6ea3fd45345a60","observation_id":"f449d21f-b2be-439f-b653-db349c7fd0fe","resolution":{"observed_at":"2026-08-02T03:56:20.714439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.799675Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.799675Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:b143abb602cd10cb008067971e84e482cf61f8ac00d178b3f17b4645bbc36c0d","observation_id":"896c379b-cc53-476c-90ac-d0aef515b035","resolution":{"observed_at":"2026-08-02T03:56:20.799675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:20.880149Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:20.880149Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:8ed13dee02f225f402d1b616828f8ba35b560ea17ee31813391e4173637bf6ef","observation_id":"3de3af29-3ee5-4466-a8be-a08728f475c2","resolution":{"observed_at":"2026-08-02T03:56:20.880149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.053384Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.053384Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:403a7187b2306fb004e777e4f8d577ad9c5233b374736839007e8501c9c42b9c","observation_id":"ce50c94b-a3a4-471f-be24-24441214db39","resolution":{"observed_at":"2026-08-02T03:56:21.053384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-02T03:56:21.137234Z","title":"arXiv preprint arXiv:2307.09288 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.137234Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6b63900b622328484759fed0e20d91a77357765eb29b8d582b1777b177e04bea","observation_id":"7c694cdb-d3f0-43dd-8de0-b2ad25436cbf","resolution":{"observed_at":"2026-08-02T03:56:21.137234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.226495Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.226495Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:badefe360c1ce823620598bd0f2b80a5f1d4c253d50ba3d501839ab56f7899a2","observation_id":"a7dfab01-8fb4-4bbf-80ed-fc97908c1d1b","resolution":{"observed_at":"2026-08-02T03:56:21.226495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.778799Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.778799Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:35c01ad4198b4afc6cfb54defdbd1ed7ffbe950cc1039c25ae7922d787e71d67","observation_id":"5dcdc66f-3e2e-4473-8687-5f67dabb6097","resolution":{"observed_at":"2026-08-02T03:56:21.778799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:21.914841Z","title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:21.914841Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d6306605ccfce75c1cfe4c32e0ad08ebd3145bcb56117373081ef2cbefa2e08c","observation_id":"0df156b7-8300-426f-a02a-3068a2c613f2","resolution":{"observed_at":"2026-08-02T03:56:21.914841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.005298Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.005298Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6e95cf269bbb1163f679901ca043919de99a545723d076189cabfda698beb81a","observation_id":"4a335679-0cce-4c6c-b368-e2629f9fe916","resolution":{"observed_at":"2026-08-02T03:56:22.005298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.088004Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.088004Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:f83771418c09fb1fae549848417444187b2e3a7d0be9b4f6b491af890b683b43","observation_id":"b80a5562-e38f-49c0-bb48-a47b19172bb9","resolution":{"observed_at":"2026-08-02T03:56:22.088004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.483686Z","title":"2021 , eprint=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.483686Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c2df88ce16e48bd247f433189ff7d81c4e1f89dc637bc76c8a00640ec94f5559","observation_id":"81448863-aa35-4d29-8385-7335affaafb8","resolution":{"observed_at":"2026-08-02T03:56:22.483686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-02T03:56:22.639590Z","title":"arXiv preprint arXiv:2107.03374 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.639590Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6a1f7924975b71010164d6551f20b91b048ca4f4fbc81ebe37d6e99b4e4a8cec","observation_id":"ca59b8ff-fdf7-4965-84ec-c4351306bedb","resolution":{"observed_at":"2026-08-02T03:56:22.639590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.760061Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.760061Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:608e6bc6688530c98f4e6e68551359ee611724db2a59c7efaab93d18641e9432","observation_id":"61fb4327-a01c-4cd3-a077-1fa48da83bf6","resolution":{"observed_at":"2026-08-02T03:56:22.760061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:22.894503Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:22.894503Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:fcf96b4c0095d6411bc90b34e2937d39694c9baa4c8c65b34775d38be4debd28","observation_id":"d3b39df6-4f93-4661-9ee9-3888b3ca71ef","resolution":{"observed_at":"2026-08-02T03:56:22.894503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.030533Z","title":"2025 , note=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.030533Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:92813612a434e0d6517df8825682aaab9868809bf04b19040dbf738f82f47eba","observation_id":"348de46d-1511-4261-bede-1369d2b10dc5","resolution":{"observed_at":"2026-08-02T03:56:23.030533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.120640Z","title":"Notion Blog , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.120640Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:afe14c6a15737dbc456fab76d091c038de915450185ae5f639af1a44a2d70ae9","observation_id":"f888b210-f776-43c4-b9c2-38565231cac3","resolution":{"observed_at":"2026-08-02T03:56:23.120640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.165656Z","title":"Proceedings of the Twentieth European Conference on Computer Systems , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.165656Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:e852d130e17d8e89db21c9ad792f219a3e63d98601cdca1dd374061724dc1ffe","observation_id":"83eb78a5-a572-4d8c-8c7a-c698a76b7835","resolution":{"observed_at":"2026-08-02T03:56:23.165656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21787","last_updated":"2024-12-30T19:03:24Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:57:25Z","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21787","snapshot_observed_at":"2026-08-02T03:56:23.249870Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.249870Z"},"links":{"cited_paper":"/paper/2407.21787","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:37effd080a69e0263b62f7b5c5a541bc15634ccce160fa4f9e47ea938a90bfbd","observation_id":"67ad94e3-5616-4e9f-917a-d7a4dcb50dd3","resolution":{"observed_at":"2026-08-02T03:56:23.249870Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.361339Z","title":"Suchanek, and Gaël Varoquaux","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.361339Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:071b06cb4601028ea7c2ccae4bfadbade600dcead45d3d6e0c9971e052a3d8d3","observation_id":"d5a324d4-d99d-4d29-b747-38185b5d4cd5","resolution":{"observed_at":"2026-08-02T03:56:23.361339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.440593Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.440593Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2152941178f0ee39eedc769d8969c811b02052f41725495ed457c3080fe7ec15","observation_id":"3c453e4c-d9c0-4800-880b-d144d49c2f3d","resolution":{"observed_at":"2026-08-02T03:56:23.440593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-04T15:46:25.710484Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-02T03:56:23.497995Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.497995Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1011ba6ec8eec003056832e8431426d34daf90656cc6d803669d02befd200d6b","observation_id":"4ef649a2-8b45-4b4d-8810-387b6f215161","resolution":{"observed_at":"2026-08-02T03:56:23.497995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.548141Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.548141Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:a890007d3c552e2e429f3530b8806cc72bec0f1396ce359b1d1e682793c9a41d","observation_id":"ac7bb018-069e-414f-a716-39959d97cc90","resolution":{"observed_at":"2026-08-02T03:56:23.548141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15260","last_updated":"2025-08-21T05:48:38Z","snapshot_observed_at":"2026-08-02T03:10:09.020351Z","submitted_at":"2025-08-21T05:48:38Z","title":"Deep Think with Confidence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.15260","snapshot_observed_at":"2026-08-02T03:56:23.603454Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.603454Z"},"links":{"cited_paper":"/paper/2508.15260","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:32fde163330678cbc051f53ded426d596d1cb45fab8136474a3de6d4f8b570ad","observation_id":"feb18894-f6f1-442e-921f-1624600d7218","resolution":{"observed_at":"2026-08-02T03:56:23.603454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.658276Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.658276Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:44bb05935161ee18a3a85f509c944acb459563c1c0286b837649d088d58077a4","observation_id":"3ee690ed-577e-4134-96b9-af4d7d7f4621","resolution":{"observed_at":"2026-08-02T03:56:23.658276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.714297Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.714297Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2193a0d4acf919bd62f751a88320fc55bcef43f81b8a6d791a8fcaa1d4fa71e1","observation_id":"b82e8989-5a63-4900-8338-5ac5bd326d37","resolution":{"observed_at":"2026-08-02T03:56:23.714297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.768342Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.768342Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:4d2cfeb3fe322b086b5c5fa4fdcc024bcb54f947b23e6b91746f9943a3cc20a4","observation_id":"961b300a-ac3f-40e0-908f-beb82197fc5c","resolution":{"observed_at":"2026-08-02T03:56:23.768342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:23.823431Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.823431Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:9c8da0570fe7be23166ae4983dee965ab80e153e8993ee03cd4728e70f753fad","observation_id":"fa844820-789c-4b45-98c1-a39b92167c1a","resolution":{"observed_at":"2026-08-02T03:56:23.823431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-02T03:56:23.888612Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.888612Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:4d3574a69237352784524b7afda876219ccd5457aa2f61e4afc8c62ca776a108","observation_id":"21d58873-741b-4fd8-a84f-6f553481f917","resolution":{"observed_at":"2026-08-02T03:56:23.888612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-02T03:56:23.929756Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:23.929756Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ae85a68626342bb0ca47e4ade2f99446f64196202337fe88a4e2931360263e56","observation_id":"799011ba-2eaa-4488-894b-d1c81aebb9ee","resolution":{"observed_at":"2026-08-02T03:56:23.929756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06457","last_updated":"2024-08-14T02:41:48Z","snapshot_observed_at":"2026-07-06T17:28:02.037844Z","submitted_at":"2024-02-09T15:02:56Z","title":"V-STaR: Training Verifiers for Self-Taught Reasoners","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06457","snapshot_observed_at":"2026-08-02T03:56:24.014273Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.014273Z"},"links":{"cited_paper":"/paper/2402.06457","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:265306ea1d0e3343a3daef925941fda6e80eb43d959bc13ba6cdeb7ef07f7f2d","observation_id":"0ff3e0b5-87f7-449e-b92d-40ba825e54a2","resolution":{"observed_at":"2026-08-02T03:56:24.014273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.068987Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.068987Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2933994f7ae3cbfb1cc933ed1fe6ace6678e3f06968d26998f4451ef79f5eea8","observation_id":"75be82ed-5f57-4301-9cb3-29cd980988fa","resolution":{"observed_at":"2026-08-02T03:56:24.068987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.07079","last_updated":"2026-05-22T17:34:18Z","snapshot_observed_at":"2026-07-06T22:48:13.053749Z","submitted_at":"2026-03-07T07:26:18Z","title":"Entropy-Aware On-Policy Distillation of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.07079","snapshot_observed_at":"2026-08-02T03:56:24.076911Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.076911Z"},"links":{"cited_paper":"/paper/2603.07079","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:689d815bf36584d497031cf2a652fcef04e49ea777f80e211aa793d85ed9d6ca","observation_id":"2f972ddc-0bc2-405e-989b-8f7482a4d691","resolution":{"observed_at":"2026-08-02T03:56:24.076911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.096298Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.096298Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2d490e56d39237fbe8d8ddab64aeb9726f73c3bca7e922ea2aae23db70956e59","observation_id":"03eea6e4-c882-4cbd-872f-e4a7e795d3a8","resolution":{"observed_at":"2026-08-02T03:56:24.096298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.05221","last_updated":"2022-11-21T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-07-11T22:59:39Z","title":"Language Models (Mostly) Know What They Know","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.05221","snapshot_observed_at":"2026-08-02T03:56:24.247165Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.247165Z"},"links":{"cited_paper":"/paper/2207.05221","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:52b7b314a2f7f4c569c82954fcc08afbdb946bd0f394e70747fb8e0ff730a90d","observation_id":"554be546-d273-4163-b323-fd0cd6788333","resolution":{"observed_at":"2026-08-02T03:56:24.247165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.403912Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.403912Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:fe2cd6ba0c0303aae7c821012f2183d1dd7846984edba66375a81240ad3611ab","observation_id":"afd62219-3532-4b9e-887e-6aa5939bf3d1","resolution":{"observed_at":"2026-08-02T03:56:24.403912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.09664","last_updated":"2023-04-15T12:55:45Z","snapshot_observed_at":"2026-07-06T14:53:27.667483Z","submitted_at":"2023-02-19T20:10:07Z","title":"Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.09664","snapshot_observed_at":"2026-08-02T03:56:24.521676Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.521676Z"},"links":{"cited_paper":"/paper/2302.09664","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:b82f6d93d8cd3c2aa148e1df1892f34f26512c44454450c4d4d0929ae1c5566d","observation_id":"680bbf43-3e73-46f7-a35d-adc395118d28","resolution":{"observed_at":"2026-08-02T03:56:24.521676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.14858","last_updated":"2022-07-01T02:15:12Z","snapshot_observed_at":"2026-08-05T15:41:22.691461Z","submitted_at":"2022-06-29T18:54:49Z","title":"Solving Quantitative Reasoning Problems with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.14858","snapshot_observed_at":"2026-08-02T03:56:24.673658Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.673658Z"},"links":{"cited_paper":"/paper/2206.14858","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:df2c3b9baa532b29a6c17d1106f9b3780400894f9c4500645458d6519922abf6","observation_id":"c8716189-2874-4206-8d3d-81b5277bf5fa","resolution":{"observed_at":"2026-08-02T03:56:24.673658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:24.777752Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.777752Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:62ac27c8b4e26ee70a7e87104d5e12b4ee4d33277681d6aa2c2c3bfad51ec84a","observation_id":"3920efc6-6911-4955-a8a4-d4ed1b692a53","resolution":{"observed_at":"2026-08-02T03:56:24.777752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14334","last_updated":"2022-06-13T05:04:53Z","snapshot_observed_at":"2026-08-03T04:20:54.522222Z","submitted_at":"2022-05-28T05:02:31Z","title":"Teaching Models to Express Their Uncertainty in Words","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.14334","snapshot_observed_at":"2026-08-02T03:56:24.978542Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:24.978542Z"},"links":{"cited_paper":"/paper/2205.14334","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:43ba5503ce74968773ed46711a75eabdceb3cc9a871f20e1dc07858b5ef20e53","observation_id":"8dc3cc09-663e-4bd5-aaf4-9188b42317c6","resolution":{"observed_at":"2026-08-02T03:56:24.978542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.128541Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.128541Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d75fffa1be94fde626d56252c2926428481852c52ff7ba8cc56e619e911b4103","observation_id":"d63560fb-7b10-4f43-baa7-64685e72286a","resolution":{"observed_at":"2026-08-02T03:56:25.128541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.283578Z","title":"Tang, Manan Roongta, Colin Cai, Jeffrey Luo, Tianjun Zhang, Li Erran Li, Raluca Ada Popa, and Ion Stoica","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.283578Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:508ff6af32545ade55c5828123f6c945ed372cdc2a7b2014b3adcf875eaf8547","observation_id":"4b0f8d4a-8e97-48d2-bd37-9c2575b34c86","resolution":{"observed_at":"2026-08-02T03:56:25.283578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.392473Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.392473Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:bf7ed75efb58bac4adb176e0173f361c7eb69310b05558e615b58c7894e6d0bf","observation_id":"823f7f7e-bf7b-4604-8f60-852d7fc3c112","resolution":{"observed_at":"2026-08-02T03:56:25.392473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.502679Z","title":"Cooper, and Milos Hauskrecht","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.502679Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:0f09db9cee09e8109e4ad4664e83c5b8c40739e88390aabfdb36832a6cc896fe","observation_id":"0cb65d26-ae1a-4f91-a47f-594dcfa87961","resolution":{"observed_at":"2026-08-02T03:56:25.502679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.00114","last_updated":"2021-11-30T21:32:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-11-30T21:32:46Z","title":"Show Your Work: Scratchpads for Intermediate Computation with Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.00114","snapshot_observed_at":"2026-08-02T03:56:25.601371Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.601371Z"},"links":{"cited_paper":"/paper/2112.00114","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:d57080dfb8ee3d6e8f5b48ca23ae8715993758cdfc134b51bdba91550727be6c","observation_id":"0b77afcd-db0d-4e7f-8d39-71960060d9f8","resolution":{"observed_at":"2026-08-02T03:56:25.601371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-02T03:56:25.765151Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.765151Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c866c482ab924a398bd92b2236638777530171348b836bf078feb5d912d5783e","observation_id":"ee3d3407-7d5b-4cf3-9a81-0fbaf9356a30","resolution":{"observed_at":"2026-08-02T03:56:25.765151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:25.929517Z","title":null,"venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:25.929517Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2c1626838dbf0b5655e4ce60a078d160b6c3928d618ca6ed95f4bbc38758d83f","observation_id":"8dcf0309-6377-4327-b1e6-74cf0cf49094","resolution":{"observed_at":"2026-08-02T03:56:25.929517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-02T03:56:26.085330Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.085330Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:4f8a506fe945478bb7c8ec3f7622a2617a9cf6a554f3a585435bca96612b3fa3","observation_id":"23048bec-69ee-42ce-ae5d-6f87b4149b89","resolution":{"observed_at":"2026-08-02T03:56:26.085330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-02T03:56:26.201200Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.201200Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:b142ff0ffac4517bbb797cceb80c4b3cea40fa0a5799281e4b95ac3e4c5d1f09","observation_id":"cff3b64b-e71c-4d53-a4e5-2c8f3aed78e4","resolution":{"observed_at":"2026-08-02T03:56:26.201200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-02T03:56:26.300681Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.300681Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:a632702d7e409e9b341adf77dbd89a7e7ae4706a4a08175855f821a53b50c20e","observation_id":"baca7525-b88d-4f3b-868d-4c077fe4f304","resolution":{"observed_at":"2026-08-02T03:56:26.300681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:26.472307Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.472307Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:36fc2f2168406c50accae340ce08d8a79518cc260929daadaa56cec7eb237ce1","observation_id":"77d4108a-1dbd-40c2-9bc3-66605ce49084","resolution":{"observed_at":"2026-08-02T03:56:26.472307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06585","last_updated":"2024-04-18T03:12:09Z","snapshot_observed_at":"2026-07-06T16:59:56.350207Z","submitted_at":"2023-12-11T18:17:43Z","title":"Beyond Human Data: Scaling Self-Training for Problem-Solving with Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06585","snapshot_observed_at":"2026-08-02T03:56:26.701252Z","title":"Co-Reyes, Rishabh Agarwal, Ankesh Anand, Piyush Patil, Xavier Garcia, Peter J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.701252Z"},"links":{"cited_paper":"/paper/2312.06585","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:6bfb2743506c039eecaf26923ec8dd828673c9e512f25f028fbc15e9986fdbc5","observation_id":"e0ff153d-3cab-4e56-aead-9b7e7bd48a79","resolution":{"observed_at":"2026-08-02T03:56:26.701252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-02T03:56:26.864154Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.864154Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:90a351ceeca2ca1095645e81edd32f399e26d70a362882d5c134838d6de4dab7","observation_id":"9c2de194-814c-47ec-aafd-045bfe1db7d1","resolution":{"observed_at":"2026-08-02T03:56:26.864154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:26.959184Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:26.959184Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:2853d92967fa45036a3236c443549cfebd5045f89a522cedd858d053d67a5846","observation_id":"dd6fe679-a26e-42ef-b5d6-8390dda94ef1","resolution":{"observed_at":"2026-08-02T03:56:26.959184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.043472Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.043472Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:80268fd13e21f1b00f5604d1107828b824e839ff2e2ea36549bb92a583fe3bb3","observation_id":"bfe0b2c0-1b95-40de-b0f3-e1591a58c3a4","resolution":{"observed_at":"2026-08-02T03:56:27.043472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14275","last_updated":"2022-11-25T18:19:44Z","snapshot_observed_at":"2026-08-01T02:16:43.109337Z","submitted_at":"2022-11-25T18:19:44Z","title":"Solving math word problems with process- and outcome-based feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.14275","snapshot_observed_at":"2026-08-02T03:56:27.103545Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.103545Z"},"links":{"cited_paper":"/paper/2211.14275","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:1fc1aeb97717293f6a5681af9550423e7c81ea44dd20cce1336831bfef3f8d93","observation_id":"b3d05b55-f6b1-4e03-b256-80f2afb2143a","resolution":{"observed_at":"2026-08-02T03:56:27.103545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09265","last_updated":"2025-09-11T08:50:01Z","snapshot_observed_at":"2026-08-04T19:28:55.084329Z","submitted_at":"2025-09-11T08:50:01Z","title":"Harnessing Uncertainty: Entropy-Modulated Policy Gradients for Long-Horizon LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.09265","snapshot_observed_at":"2026-08-02T03:56:27.159465Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.159465Z"},"links":{"cited_paper":"/paper/2509.09265","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:89f9a6a35d9dd09799abf9416ac2ce8b13c16eece17d2ab0f1b2043eb39b1cf1","observation_id":"42bbac1a-559f-4d9c-b285-f70dfc65cc5b","resolution":{"observed_at":"2026-08-02T03:56:27.159465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.235699Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.235699Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:0a58dc2655f19666e489e0b2cbd696d4e3312e56e768d88b5a90ed5c64d2250b","observation_id":"18fd48cc-3fa1-4ebe-ba9b-d3f7a56333c9","resolution":{"observed_at":"2026-08-02T03:56:27.235699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01939","last_updated":"2025-11-13T10:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T17:54:39Z","title":"Beyond the 80/20 Rule: High-Entropy Minority Tokens Drive Effective Reinforcement Learning for LLM Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01939","snapshot_observed_at":"2026-08-02T03:56:27.321716Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.321716Z"},"links":{"cited_paper":"/paper/2506.01939","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:c4427b0cdc429c4940890ee6790e581019197009f81f74f76c8ca9bd1b31bf91","observation_id":"1e9aa211-3d85-4e7e-b7ee-9e22139d5307","resolution":{"observed_at":"2026-08-02T03:56:27.321716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-02T03:56:27.406144Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.406144Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:cc2fcf27cae2fbf60afdebe078c6ceeb067c11976228ad023127db8cfff6a9f9","observation_id":"aee81d70-fd58-468c-b624-0423198e0d1b","resolution":{"observed_at":"2026-08-02T03:56:27.406144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.00487","last_updated":"2026-05-30T02:39:40Z","snapshot_observed_at":"2026-08-02T08:25:04.138756Z","submitted_at":"2026-05-30T02:39:40Z","title":"TAPS: Target-Aware Prefix Tree Selection for Diffusion-Drafted Speculative Decoding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.00487","snapshot_observed_at":"2026-08-02T03:56:27.461043Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.461043Z"},"links":{"cited_paper":"/paper/2606.00487","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:b07090ab67b7bc3406507f2f5db256e588b7242b5aa8f6f59f6ee705326eb9bf","observation_id":"895e38b0-66f5-49e2-89de-35a5e31d6882","resolution":{"observed_at":"2026-08-02T03:56:27.461043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-02T03:56:27.545672Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.545672Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:837d74df889b5bb1dc4259f940a4d0ca75e98d1e5c55b01420653bb70b6cf9f9","observation_id":"f6b27116-e6a6-4ab6-8f5c-b9a11e92cbd0","resolution":{"observed_at":"2026-08-02T03:56:27.545672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-08-02T03:56:27.628364Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.628364Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:329601dca55dd2c3b94785740f9ce90205e7ea3bd8cb350ae2cbb8bb717010cf","observation_id":"0cba5a80-bd82-4469-ab58-fa7c906b6ecb","resolution":{"observed_at":"2026-08-02T03:56:27.628364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.647136Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.647136Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:f4e947ebfcc920d15e869bd7324d5b48a407888a28273be5a5215e1ef46f0bc1","observation_id":"67dd4cdc-1227-43f2-989e-788c9c1f206f","resolution":{"observed_at":"2026-08-02T03:56:27.647136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-02T03:56:27.724507Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.724507Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:ce340d32fe88eafcb736187b432017918ebc137b00da6d9dccecf427b34ebc1e","observation_id":"7e058d93-dbaa-4d53-9c0c-4d9c58af3a61","resolution":{"observed_at":"2026-08-02T03:56:27.724507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-07-06T19:17:41.512834Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-02T03:56:27.820539Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.820539Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:af069a4f7cfda84143c2252a0e0a176ec3e04c54209ac66134a8760b54dbd3ce","observation_id":"2a416eac-6665-4247-b66c-dc64f2311023","resolution":{"observed_at":"2026-08-02T03:56:27.820539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:27.965607Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:27.965607Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:29767e4416596ad9d6a0d90a76c56284c5784f07bc20ff51f12cf6caf330a489","observation_id":"db62ca64-039c-4e18-bb53-303a7ae71964","resolution":{"observed_at":"2026-08-02T03:56:27.965607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T03:56:28.073779Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration","version":2},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-08-02T03:56:28.073779Z"},"links":{"citing_paper":"/paper/2607.13753"},"observation_digest":"sha256:fa717fcfae143a3de9f43d0acdf43eaf42fc8ff9400cc8bceec13193dcb1be8c","observation_id":"1df684f4-74c3-4f48-b436-f3abc4622833","resolution":{"observed_at":"2026-08-02T03:56:28.073779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.13753","last_updated":"2026-07-20T09:24:51Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-02T03:56:15.228598Z","submitted_at":"2026-07-15T12:12:37Z","title":"Post-Training Shifts Confidence: A Three-Stage Analysis of How SFT, RL, and OPD Shape CoT Calibration"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":100,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":108},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 100 of 108 outbound references and 0 inbound Pith citation observations for arXiv:2607.13753."}