{"as_of":"2026-08-10T17:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7b558d675e13def27ffbb8eb985c8fcdfd125a1158993788a8a078fac561118e","coverage":[{"denominator":101,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:20:17.762309Z","state":"measured"},{"denominator":106,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":106,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T23:52:36.891080Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"cited_work":{"arxiv_id":"2507.16331","doi":"10.48550/arxiv.2507.16331","metadata_source":"arxiv_reference","pith_arxiv_id":"2507.16331","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xunjian Yin, Xinyi Wang, Liangming Pan, Li Lin, Xiaojun Wan, and William Yang Wang","venue":"ArXiv.org","work_id":"fd1fc939-0d26-4c4a-8f64-6c365c688c31","year":2025},"citing_paper":{"arxiv_id":"2512.13399","last_updated":"2026-05-13T04:43:51Z","snapshot_observed_at":"2026-08-02T20:35:00.137172Z","submitted_at":"2025-12-15T14:50:08Z","title":"Differentiable Evolutionary Reinforcement Learning","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T22:34:47.984401Z"},"links":{"cited_paper":"/paper/2507.16331","citing_paper":"/paper/2512.13399"},"observation_digest":"sha256:90f5f018a7d0742978fe4c20226d867f0d8820e9815315ca290a7fa7f87e1cf8","observation_id":"8c4c22c5-3eba-4d9f-b069-19726470732d","resolution":{"observed_at":"2026-07-07T02:15:55.820429Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"cited_work":{"arxiv_id":"2507.16331","doi":"10.48550/arxiv.2507.16331","metadata_source":"arxiv_reference","pith_arxiv_id":"2507.16331","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xunjian Yin, Xinyi Wang, Liangming Pan, Li Lin, Xiaojun Wan, and William Yang Wang","venue":"ArXiv.org","work_id":"fd1fc939-0d26-4c4a-8f64-6c365c688c31","year":2025},"citing_paper":{"arxiv_id":"2604.05820","last_updated":"2026-07-14T09:49:12Z","snapshot_observed_at":"2026-07-17T23:18:01.233145Z","submitted_at":"2026-04-07T12:53:42Z","title":"SpecRL: Reinforcement Learning with Test-Based Completeness Rewards for Formal Specification Synthesis","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T19:06:10.718673Z"},"links":{"cited_paper":"/paper/2507.16331","citing_paper":"/paper/2604.05820"},"observation_digest":"sha256:b9139f4a34d0e5849d620e25bd5f8805a248125069f4c9704b9a0ff6d665e47e","observation_id":"bbe7d2f0-f928-4619-8b21-8e28f7b5c114","resolution":{"observed_at":"2026-07-07T02:15:55.820429Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"cited_work":{"arxiv_id":"2507.16331","doi":"10.48550/arxiv.2507.16331","metadata_source":"arxiv_reference","pith_arxiv_id":"2507.16331","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xunjian Yin, Xinyi Wang, Liangming Pan, Li Lin, Xiaojun Wan, and William Yang Wang","venue":"ArXiv.org","work_id":"fd1fc939-0d26-4c4a-8f64-6c365c688c31","year":2025},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":1},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-05-07T06:30:09.945371Z"},"links":{"cited_paper":"/paper/2507.16331","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:894eea4244bb58f00046e4eff7ae40feb0366a9ee546018ee665650b8bf0a721","observation_id":"bc463633-a4ff-4710-a2b0-84d885712431","resolution":{"observed_at":"2026-07-07T02:15:55.820429Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"cited_work":{"arxiv_id":"2507.16331","doi":"10.48550/arxiv.2507.16331","metadata_source":"arxiv_reference","pith_arxiv_id":"2507.16331","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xunjian Yin, Xinyi Wang, Liangming Pan, Li Lin, Xiaojun Wan, and William Yang Wang","venue":"ArXiv.org","work_id":"fd1fc939-0d26-4c4a-8f64-6c365c688c31","year":2025},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":2},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-05-08T03:12:19.414358Z"},"links":{"cited_paper":"/paper/2507.16331","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:9cfd2f9a5d0c339cfabb5138d692ca5183242d8417bffc418244e02b81c62b1d","observation_id":"7f4adce5-b04a-4521-8593-4de917f2e2e4","resolution":{"observed_at":"2026-07-07T02:15:55.820429Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"cited_work":{"arxiv_id":"2507.16331","doi":"10.48550/arxiv.2507.16331","metadata_source":"arxiv_reference","pith_arxiv_id":"2507.16331","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xunjian Yin, Xinyi Wang, Liangming Pan, Li Lin, Xiaojun Wan, and William Yang Wang","venue":"ArXiv.org","work_id":"fd1fc939-0d26-4c4a-8f64-6c365c688c31","year":2025},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":3},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-05-19T16:58:41.558250Z"},"links":{"cited_paper":"/paper/2507.16331","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:6ca798a1ce7319a34188f6ff39e889dc4c4cd11b2317ee15e1660f83e31f9e4a","observation_id":"727b29f7-7158-48e6-8be7-2c89ade6de60","resolution":{"observed_at":"2026-07-07T02:15:55.820429Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"cited_work":{"arxiv_id":"2507.16331","doi":"10.48550/arxiv.2507.16331","metadata_source":"arxiv_reference","pith_arxiv_id":"2507.16331","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xunjian Yin, Xinyi Wang, Liangming Pan, Li Lin, Xiaojun Wan, and William Yang Wang","venue":"ArXiv.org","work_id":"fd1fc939-0d26-4c4a-8f64-6c365c688c31","year":2025},"citing_paper":{"arxiv_id":"2605.30914","last_updated":"2026-05-29T06:59:28Z","snapshot_observed_at":"2026-08-04T08:48:12.582835Z","submitted_at":"2026-05-29T06:59:28Z","title":"Automating Formal Verification with Reinforcement Learning and Recursive Inference","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-06-28T23:52:36.891080Z"},"links":{"cited_paper":"/paper/2507.16331","citing_paper":"/paper/2605.30914"},"observation_digest":"sha256:3e4d6ac292517de0ff721b2c7780c7e35e259576c1fe42385551c97179f10b6f","observation_id":"e04202e0-9fbc-44b0-8f39-e67bafc14397","resolution":{"observed_at":"2026-07-07T02:15:55.820429Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.16331/citation-record","integrity":"/paper/2507.16331/integrity","json":"/paper/2507.16331/citation-record.json","paper":"/paper/2507.16331"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:10.030154Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.030154Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:a16fb801d22bc6b5be4452817a8a7752f2ca05be0a93cb2ec0785ba8c9d5ba0a","observation_id":"209662fa-e112-499d-9859-5b1b9b67a6a6","resolution":{"observed_at":"2026-08-06T15:20:10.030154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T15:20:10.126663Z","title":"L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.126663Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:44fb226e2d2884a7534056bdfc0f902cbf6c830f9a3bda61be8fe46a80c5ecd0","observation_id":"2c44f09e-db6c-45a8-811d-c2381f389bfd","resolution":{"observed_at":"2026-08-06T15:20:10.126663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:10.230005Z","title":"Alphacode 2 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.230005Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:652d73b88b3d79bf6f7549dd8b3049778bd06ceb6b09a5c17095c648a3dafe18","observation_id":"02566cd0-1fa5-4b59-82c4-35ebb3cdd1b0","resolution":{"observed_at":"2026-08-06T15:20:10.230005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:10.313716Z","title":"System card: Claude opus 4 & claude sonnet 4","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.313716Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:3bd65e233f872ea69d2d596bd62bc5225c0624f127fa14afc54ee6b731a19295","observation_id":"f9a21c30-57a9-43b6-8cad-fc0c63416986","resolution":{"observed_at":"2026-08-06T15:20:10.313716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-06T15:20:10.412132Z","title":"Program synthesis with large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.412132Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:15e6b5a7460595f194bdc948c327d42e27db9c507a564414f62abaeb0d31f095","observation_id":"795fb0d1-7bd8-4456-a8c7-a4c6cd8553db","resolution":{"observed_at":"2026-08-06T15:20:10.412132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:10.494851Z","title":"Y., Collignon, N., Neo, C., Lee, I., Paren, A., Bibi, A., Trager, R., Fornasiere, D., Yan, J., Elazar, Y., and Bengio, Y","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.494851Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:e3e0ed4c87e7d535571691a2c0fa610983ba38f341d54851a5f65e829811c683","observation_id":"e0412022-7c59-4522-98d8-4a53737f451f","resolution":{"observed_at":"2026-08-06T15:20:10.494851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-06T15:20:10.590600Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.590600Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:494cbaba82c1382185cb31a6c3cc29fde6e0c3677c6f7567a7b166b7d0aa2a05","observation_id":"ac5644be-2809-4111-9edf-db13e3301b34","resolution":{"observed_at":"2026-08-06T15:20:10.590600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-08-08T22:33:20.124926Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09567","snapshot_observed_at":"2026-08-06T15:20:10.705035Z","title":"Towards reasoning era: A survey of long chain-of-thought for reasoning large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.705035Z"},"links":{"cited_paper":"/paper/2503.09567","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:f774bb4b8755ac9e3f529a5976e881477e12330b832b203b45574489d1a2d92f","observation_id":"60a789b0-35ac-4a97-b03a-25a13e8eb4b8","resolution":{"observed_at":"2026-08-06T15:20:10.705035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05410","last_updated":"2025-05-08T16:51:43Z","snapshot_observed_at":"2026-07-06T21:21:01.070459Z","submitted_at":"2025-05-08T16:51:43Z","title":"Reasoning Models Don't Always Say What They Think","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.05410","snapshot_observed_at":"2026-08-06T15:20:10.865469Z","title":"Reasoning models don't always say what they think","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.865469Z"},"links":{"cited_paper":"/paper/2505.05410","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:cc31935c0018ba7e27384b594d9211ab61c243d9c5756a185fa88335022accca","observation_id":"5795bd41-e36f-4944-a6bc-19e2b558e834","resolution":{"observed_at":"2026-08-06T15:20:10.865469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:10.997966Z","title":"A., Nielsen-Garcia, C., Mir, S., Li, S., Orender, J., et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:10.997966Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:bc72eee8c54d738d2d8cc81fc8947fe7e4cc593240f410152ddebb57b988f3d0","observation_id":"90753e9c-6ab2-4c7c-a713-2cbb39979498","resolution":{"observed_at":"2026-08-06T15:20:10.997966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.01547","last_updated":"2019-11-25T13:02:04Z","snapshot_observed_at":"2026-07-06T08:34:41.399203Z","submitted_at":"2019-11-05T00:31:38Z","title":"On the Measure of Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.01547","snapshot_observed_at":"2026-08-06T15:20:11.175293Z","title":"On the measure of intelligence","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.175293Z"},"links":{"cited_paper":"/paper/1911.01547","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:aabe95427529742f2006e5de0916b0c041f42115f98789b3af66ebc30f8d2dfd","observation_id":"25498567-1427-4997-97e9-f22990e8e5a0","resolution":{"observed_at":"2026-08-06T15:20:11.175293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:11.307777Z","title":"V., Levine, S., and Ma, Y","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.307777Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:bcf04c435513e34aabc75d6846cb91bc58240b20198fefb2d297ee8775ab75f1","observation_id":"26e30546-1922-4ffc-8ca8-ef287dabe3bf","resolution":{"observed_at":"2026-08-06T15:20:11.307777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:11.457296Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.457296Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:e6f7c54fc8f5fe7db1f3a931b1366ec06137272a05d475167fc59986fb803858","observation_id":"d1e27b7a-75f7-4d4e-b5d7-77d471bc4b2b","resolution":{"observed_at":"2026-08-06T15:20:11.457296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:11.623988Z","title":"Towards formal verification of llm-generated code from natural language prompts, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.623988Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:326016a6b92edaf38e394ed54aacf15dee04838c4a4601418b0b7cd6349f669e","observation_id":"2be1b238-97bc-4567-b15c-b40184d03117","resolution":{"observed_at":"2026-08-06T15:20:11.623988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.06624","last_updated":"2024-07-08T13:35:00Z","snapshot_observed_at":"2026-08-10T07:42:23.181616Z","submitted_at":"2024-05-10T17:38:32Z","title":"Towards Guaranteed Safe AI: A Framework for Ensuring Robust and Reliable AI Systems","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.06624","snapshot_observed_at":"2026-08-06T15:20:11.745072Z","title":"Towards guaranteed safe ai: A framework for ensuring robust and reliable ai systems","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.745072Z"},"links":{"cited_paper":"/paper/2405.06624","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:3cda558545181676fcdd60a148679fda489d4b6dffb66b351ddf4eea089d0d5f","observation_id":"4991e340-828a-47c7-89e5-c1ea9f56ce2c","resolution":{"observed_at":"2026-08-06T15:20:11.745072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:11.837567Z","title":"and Bj rner, N","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.837567Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:d59b240ed738773d624b06ef9ca0cd10fb5423cc5dde29958c989583ea84663b","observation_id":"e6b136cd-1e6d-4670-a8ac-6019bbc6c03e","resolution":{"observed_at":"2026-08-06T15:20:11.837567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:11.938086Z","title":"The lean theorem prover (system description)","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.938086Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:72a913fc7b284b8a53938101c276e68697bb3faae83b34a237954e4c1575be22","observation_id":"3ec856cf-374a-452e-a71e-5e62e3ef5587","resolution":{"observed_at":"2026-08-06T15:20:11.938086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.21801","last_updated":"2025-07-18T08:20:23Z","snapshot_observed_at":"2026-08-02T11:46:58.603316Z","submitted_at":"2025-04-30T16:57:48Z","title":"DeepSeek-Prover-V2: Advancing Formal Mathematical Reasoning via Reinforcement Learning for Subgoal Decomposition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.21801","snapshot_observed_at":"2026-08-06T15:20:11.998704Z","title":"Deepseek-prover-v2: Advancing formal mathematical reasoning via reinforcement learning for subgoal decomposition","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:11.998704Z"},"links":{"cited_paper":"/paper/2504.21801","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:b21a803b2389684af64f74186ea00ca1a0329fcc78a2b2548cde98734ee90141","observation_id":"d6411b08-26cb-4ce9-abd7-6283c52494b4","resolution":{"observed_at":"2026-08-06T15:20:11.998704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1460.28915","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.315176Z","title":null,"venue":null,"work_id":"85b3dd70-11b2-4f62-8087-6d436f151f5f","year":1979},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.089275Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:a2cad31a5a5bbaae5e1f3b0c31c04f650a3adc3b1d687b62f34e4fabc70b7e95","observation_id":"b52ef171-fe8c-4273-8e73-1971028cba1b","resolution":{"observed_at":"2026-08-06T15:20:19.326180Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:12.171055Z","title":"Generalization or memorization: Data contamination and trustworthy evaluation for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.171055Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:5f3ebea67ccfa38340696432432c7482fc63203c1d8df3f33dc0848177a38d80","observation_id":"01c4464f-5ba6-4ec1-8495-06d50c2f1192","resolution":{"observed_at":"2026-08-06T15:20:12.171055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:12.270776Z","title":"and Mehta, R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.270776Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:98575d7a42a4da95d64ff1a0614f77789f3f3e7b61c4fa5acb3795e4f1a69f20","observation_id":"4fd3d24a-fa49-4e37-9c26-61b12410f04b","resolution":{"observed_at":"2026-08-06T15:20:12.270776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:12.349174Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.349174Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:fc2e5199bc0bbf7bc25eb2794de03c5f47a329953eb4e82fd8b2a678f51c5cc3","observation_id":"cec73c0b-d6f1-418d-9a79-2494f8f60d25","resolution":{"observed_at":"2026-08-06T15:20:12.349174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T15:20:12.410202Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.410202Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:a840aff41a8e95adcad9d22368637d39ecd056b6fb78c34b5e25b9500b6cc58e","observation_id":"ea85e27b-028d-485e-9367-e23dd18282df","resolution":{"observed_at":"2026-08-06T15:20:12.410202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2022.findings-emnlp.66","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Measuring and improving semantic diversity of dialogue generation","venue":null,"work_id":"e6bef3c3-d2b0-4c7c-ad1e-a52c36e2ac10","year":2022},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.504634Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:886e978bc4ae6206da2f10d9f1b5880e913fc9e83e1ef5016c558d9528d1657f","observation_id":"64a92f93-6be1-4ee8-869d-ae8a5868953f","resolution":{"observed_at":"2026-08-06T15:20:17.829554Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10452","last_updated":"2025-05-29T13:17:33Z","snapshot_observed_at":"2026-08-07T17:06:48.275270Z","submitted_at":"2025-03-13T15:18:56Z","title":"DynaCode: A Dynamic Complexity-Aware Code Benchmark for Evaluating Large Language Models in Code Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10452","snapshot_observed_at":"2026-08-06T15:20:12.595812Z","title":"Dynacode: A dynamic complexity-aware code benchmark for evaluating large language models in code generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.595812Z"},"links":{"cited_paper":"/paper/2503.10452","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:044ffc4d5ede0051059f17619f5d6998fe90c277954d9b30c69868a39fb5b6c3","observation_id":"6b15fe36-310d-49e9-b19c-5aa8163d853c","resolution":{"observed_at":"2026-08-06T15:20:12.595812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:12.662973Z","title":"Does math reasoning improve general llm capabilities? understanding transferability of llm reasoning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.662973Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:e1bcf223538059dbfcfb93eb80c253e4f6e3181ea5f80eeba06d229e63870a13","observation_id":"b2fdcece-c286-43c5-ae3f-083c6e84d34e","resolution":{"observed_at":"2026-08-06T15:20:12.662973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-06T15:20:12.732299Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.732299Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:8a6aa4940dd228ddb22f7ccd943a2f2603941818cbfa2209243dbefbf15d913d","observation_id":"30ba6c9b-bfd9-4a36-ad2f-bae3d0d4eb87","resolution":{"observed_at":"2026-08-06T15:20:12.732299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.09192","last_updated":"2025-05-27T17:24:38Z","snapshot_observed_at":"2026-08-08T23:01:32.478617Z","submitted_at":"2025-02-13T11:32:09Z","title":"Thinking beyond the anthropomorphic paradigm benefits LLM research","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.09192","snapshot_observed_at":"2026-08-06T15:20:12.807230Z","title":"and Cheng, M","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.807230Z"},"links":{"cited_paper":"/paper/2502.09192","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:9e697ad12677f4bdd148967b3ee347307374023af085845e3cb799fc58890438","observation_id":"05843a20-46b4-47a9-95ab-6461b612471b","resolution":{"observed_at":"2026-08-06T15:20:12.807230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:12.897446Z","title":"AI safety via debate, May 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.897446Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:17e4c609023bc2dfddd4ba746112a4eb15b9f361cceae2c5ad806b2c9655347b","observation_id":"53afef7e-0c5c-4a08-86ee-cc062f8b3c81","resolution":{"observed_at":"2026-08-06T15:20:12.897446Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:12.967941Z","title":"Verifast: A powerful, sound, predictable, fast verifier for c and java","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:12.967941Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:dde35481e2240cf316dab0cccfb2182ee77eefde3696e89bc0bdfbc942284f3c","observation_id":"b7325784-33cd-4604-be45-4593b05e499b","resolution":{"observed_at":"2026-08-06T15:20:12.967941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:13.057170Z","title":"Do we need to verify step by step? rethinking process supervision from a theoretical perspective, February 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.057170Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:ac6134173f20a04c819ebc849fc13140a34173c1ebc24eabef981c9e125b62cb","observation_id":"a749aac7-cb21-403c-88e8-a6161cebdcde","resolution":{"observed_at":"2026-08-06T15:20:13.057170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:13.131215Z","title":"Can large language models understand intermediate representations in compilers?, February 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.131215Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:9285058694d22eff8d6fca76a8d2f9f76b7eb15eecaffd6b4a06b82d71f17beb","observation_id":"48dfe5ee-5ce7-48f6-818c-5b95c39e1955","resolution":{"observed_at":"2026-08-06T15:20:13.131215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:13.225730Z","title":"sel4: Formal verification of an os kernel","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.225730Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:bcd10be1f49b3617c8b5ca1d4f2f5e36df0d6b1b18a3ae12c4ea505f7808c4bb","observation_id":"aba1d025-7b60-4c73-933a-57588e033ef2","resolution":{"observed_at":"2026-08-06T15:20:13.225730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:13.291719Z","title":"Chain of thought monitorability: A new and fragile opportunity for AI safety, July 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.291719Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:3b4e6212e210b7911ac4813504eb4d8430a3210c21006b38e9b5394f9b319db8","observation_id":"8a58bc26-c624-463f-be6e-c5bc3fe1f6ab","resolution":{"observed_at":"2026-08-06T15:20:13.291719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:13.384938Z","title":"Gradual disempowerment: Systemic existential risks from incremental AI development, January 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.384938Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:b3af4408bfaae299d21fba9d4e4cd4a621330b53b49a0ca0f49ab6da681c2ee7","observation_id":"99a75dab-f5f2-430e-973f-fbec1f26a294","resolution":{"observed_at":"2026-08-06T15:20:13.384938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21321","last_updated":"2025-03-24T09:34:38Z","snapshot_observed_at":"2026-08-09T09:18:08.427943Z","submitted_at":"2025-02-28T18:59:54Z","title":"LLM Post-Training: A Deep Dive into Reasoning Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.21321","snapshot_observed_at":"2026-08-06T15:20:13.473496Z","title":"M., Cholakkal, H., Shah, M., Yang, M.-H., Torr, P","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.473496Z"},"links":{"cited_paper":"/paper/2502.21321","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:849b4428a2d1ca4e6d2032445d92617045d4e3cc918d495b8b9df8ace7fc224a","observation_id":"dc700f65-d7bd-49ec-895f-0a9c1db2e5c5","resolution":{"observed_at":"2026-08-06T15:20:13.473496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13702","last_updated":"2023-07-17T01:08:39Z","snapshot_observed_at":"2026-07-31T04:21:27.501709Z","submitted_at":"2023-07-17T01:08:39Z","title":"Measuring Faithfulness in Chain-of-Thought Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13702","snapshot_observed_at":"2026-08-06T15:20:13.565564Z","title":"Measuring faithfulness in chain-of-thought reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.565564Z"},"links":{"cited_paper":"/paper/2307.13702","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:a0f763d1898ae45b1eec41397f2ef74e03708f47d52b5c2f537fb20fa0e2eb45","observation_id":"ffd8c634-6d62-4e5f-aa76-105e55296bda","resolution":{"observed_at":"2026-08-06T15:20:13.565564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.01780","last_updated":"2022-11-03T08:32:59Z","snapshot_observed_at":"2026-08-10T05:41:57.950959Z","submitted_at":"2022-07-05T02:42:15Z","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.01780","snapshot_observed_at":"2026-08-06T15:20:13.692122Z","title":"D., Savarese, S., and Hoi, S","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.692122Z"},"links":{"cited_paper":"/paper/2207.01780","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:f8f05f250392ffd2b9b8e8f6e982f13e6eb719c8672cc64ddad43440434f5e66","observation_id":"e0b3a6a7-e2f6-475e-8f4c-d92da744b0b1","resolution":{"observed_at":"2026-08-06T15:20:13.692122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01141","last_updated":"2025-04-01T00:41:36Z","snapshot_observed_at":"2026-08-08T04:02:43.828729Z","submitted_at":"2025-03-03T03:48:20Z","title":"How Well do LLMs Compress Their Own Chain-of-Thought? A Token Complexity Approach","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01141","snapshot_observed_at":"2026-08-06T15:20:13.782725Z","title":"How well do llms compress their own chain-of-thought? a token complexity approach","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.782725Z"},"links":{"cited_paper":"/paper/2503.01141","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:ca6373e0b9b95e94baa1599afcab392b52c991c5457da9f130ca7f09842426fb","observation_id":"574e74cd-6a1e-41ad-b3c1-a22270773492","resolution":{"observed_at":"2026-08-06T15:20:13.782725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:13.858775Z","title":null,"venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.858775Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:205f9610c5aba6a5506945ad0aa7eb2c9f6a63e1679268c1947ec36b5322d016","observation_id":"9683c76b-3eaf-4dfa-8559-71a99d5006ac","resolution":{"observed_at":"2026-08-06T15:20:13.858775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07316","last_updated":"2025-05-21T13:38:27Z","snapshot_observed_at":"2026-08-10T03:48:50.081958Z","submitted_at":"2025-02-11T07:26:50Z","title":"CodeI/O: Condensing Reasoning Patterns via Code Input-Output Prediction","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.07316","snapshot_observed_at":"2026-08-06T15:20:13.911084Z","title":"Codei/o: Condensing reasoning patterns via code input-output prediction","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:13.911084Z"},"links":{"cited_paper":"/paper/2502.07316","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:a56ff531ba6b0248c9f164f83ad80babeca8ec23229fd1b071283f49257fd65e","observation_id":"c368a4e0-0e10-4287-af15-2a3acd3ddee6","resolution":{"observed_at":"2026-08-06T15:20:13.911084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.05687","last_updated":"2025-07-08T05:38:24Z","snapshot_observed_at":"2026-08-09T23:22:21.217202Z","submitted_at":"2025-07-08T05:38:24Z","title":"AutoTriton: Automatic Triton Programming with Reinforcement Learning in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.05687","snapshot_observed_at":"2026-08-06T15:20:14.003355Z","title":"Autotriton: Automatic triton programming with reinforcement learning in llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.003355Z"},"links":{"cited_paper":"/paper/2507.05687","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:1ee478ea0821deaf7c232d42914f456e5656ee80afe61f28df6b6a74cd5485b5","observation_id":"87b53095-17e4-4de2-bed9-d9440d4b3123","resolution":{"observed_at":"2026-08-06T15:20:14.003355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02272","last_updated":"2024-12-02T12:36:30Z","snapshot_observed_at":"2026-07-06T19:44:55.644262Z","submitted_at":"2024-11-04T17:03:55Z","title":"Combining Induction and Transduction for Abstract Reasoning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02272","snapshot_observed_at":"2026-08-06T15:20:14.124136Z","title":"M., Tang, H., Naim, M., Nguyen, D., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.124136Z"},"links":{"cited_paper":"/paper/2411.02272","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:c6f9481c51ed08e5810018c81851aaa0bd2f8a9ec9e11773f1c97f1666e5dc64","observation_id":"daf5c64a-e53e-4567-a80d-e07a591b51a5","resolution":{"observed_at":"2026-08-06T15:20:14.124136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:14.214324Z","title":"Competition-level code generation with alphacode","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.214324Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:00c1de6124297c3e7ef13731df7f790d126b633e653b039b003df3a10f5d2a4c","observation_id":"6f838c0b-5ad9-4e72-8eac-b49769b0d00e","resolution":{"observed_at":"2026-08-06T15:20:14.214324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.06283","last_updated":"2025-01-10T17:23:14Z","snapshot_observed_at":"2026-07-06T20:19:30.457335Z","submitted_at":"2025-01-10T17:23:14Z","title":"Dafny as Verification-Aware Intermediate Language for Code Generation","version":1},"cited_work":{"arxiv_id":"2501.06283","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.06283","snapshot_observed_at":"2026-08-06T15:20:18.993538Z","title":"Dafny as Verification-Aware Intermediate Language for Code Generation","venue":"cs.SE","work_id":"e72c518b-59d9-4823-b2c9-0416c4041f7d","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.285945Z"},"links":{"cited_paper":"/paper/2501.06283","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:253b353727d75e0160b638985547ee9105347911fcefab4b91861e7bc7b32ad6","observation_id":"f062fcf9-6ca4-48f0-9a4f-5a4f8255694b","resolution":{"observed_at":"2026-08-06T15:20:18.998311Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06804","last_updated":"2025-07-07T22:38:49Z","snapshot_observed_at":"2026-08-10T00:04:33.789668Z","submitted_at":"2025-07-07T22:38:49Z","title":"Towards Solving More Challenging IMO Problems via Decoupled Reasoning and Proving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06804","snapshot_observed_at":"2026-08-06T15:20:14.369135Z","title":"Towards solving more challenging imo problems via decoupled reasoning and proving","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.369135Z"},"links":{"cited_paper":"/paper/2507.06804","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:0e258e44b9cf5814f2b60fd9a8509ccddd97e7baaf28d5e09af9355c80924ed6","observation_id":"e600c9f4-9753-41e5-87cb-0070d7103c13","resolution":{"observed_at":"2026-08-06T15:20:14.369135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:20.017484Z","title":"Goedel-prover-v2: The strongest open-source theorem prover to date, 2025","venue":null,"work_id":"1caaee5e-f6b3-42bf-a4a1-9dbfb3cc3ff4","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.440664Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:cd77e6a741695e73623528bb5e0220b25592724095e8960496ea9ec9141247f5","observation_id":"2c89db8b-9dd5-4694-8a5d-0d049739b0ad","resolution":{"observed_at":"2026-08-06T15:20:20.022233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04592","last_updated":"2025-06-05T03:16:08Z","snapshot_observed_at":"2026-08-09T04:20:47.756843Z","submitted_at":"2025-06-05T03:16:08Z","title":"Safe: Enhancing Mathematical Reasoning in Large Language Models via Retrospective Step-aware Formal Verification","version":1},"cited_work":{"arxiv_id":"2506.04592","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.04592","snapshot_observed_at":"2026-08-06T15:20:18.946831Z","title":"Safe: Enhancing Mathematical Reasoning in Large Language Models via Retrospective Step-aware Formal Verification","venue":"cs.CL","work_id":"63acbaa4-b53c-4a19-b5aa-ad99e4cde664","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.514001Z"},"links":{"cited_paper":"/paper/2506.04592","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:e69d35569886932669c416e8e9804b71208e026bd68cc5dcd3859b6e48d7b22e","observation_id":"44aa9e5a-8ce3-4ece-b7bd-f98cf55671ea","resolution":{"observed_at":"2026-08-06T15:20:18.953748Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24864","last_updated":"2025-05-30T17:59:01Z","snapshot_observed_at":"2026-08-07T07:17:16.299213Z","submitted_at":"2025-05-30T17:59:01Z","title":"ProRL: Prolonged Reinforcement Learning Expands Reasoning Boundaries in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.24864","snapshot_observed_at":"2026-08-06T15:20:14.585900Z","title":"Prorl: Prolonged reinforcement learning expands reasoning boundaries in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.585900Z"},"links":{"cited_paper":"/paper/2505.24864","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:06fcc0e96a3f926a14393dbcdec93a6ebc8ce8b1afa489b19901bf29a7ad63e0","observation_id":"1cd65a90-dbba-4744-bbdb-cdc1931fba4c","resolution":{"observed_at":"2026-08-06T15:20:14.585900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08467","last_updated":"2024-06-12T17:53:31Z","snapshot_observed_at":"2026-07-06T18:29:49.284504Z","submitted_at":"2024-06-12T17:53:31Z","title":"DafnyBench: A Benchmark for Formal Software Verification","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08467","snapshot_observed_at":"2026-08-06T15:20:14.678411Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.678411Z"},"links":{"cited_paper":"/paper/2406.08467","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:e28a4392b3f3f10626b1b004735ca951a5e16c082432304503962f7ce0053025","observation_id":"6ede6bfb-c111-489e-b696-e7aef64b5e20","resolution":{"observed_at":"2026-08-06T15:20:14.678411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19173","last_updated":"2024-02-29T13:53:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-29T13:53:35Z","title":"StarCoder 2 and The Stack v2: The Next Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19173","snapshot_observed_at":"2026-08-06T15:20:14.747793Z","title":"B., Cassano, F., Lamy-Poirier, J., Tazi, N., Tang, A., Pykhtar, D., Liu, J., Wei, Y., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.747793Z"},"links":{"cited_paper":"/paper/2402.19173","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:dcb277ea3546e1c9029ead57f186382d891f64338ec7053cbcbe31f63b841385","observation_id":"89ced50b-bbb4-4dcf-a20a-0d56e00a13c6","resolution":{"observed_at":"2026-08-06T15:20:14.747793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:20.002299Z","title":"Reasoning models can be effective without thinking, April 2025","venue":null,"work_id":"9c4092a8-6cef-439a-bb34-0ba28bbb9192","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.839523Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:39296ba7ad93679ac6fb9819e4b4c6b33be9f2811a879507eb52da1f5f93b2a8","observation_id":"5c16fc10-6966-4293-b7ba-25e84c98c600","resolution":{"observed_at":"2026-08-06T15:20:20.007260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21521","last_updated":"2025-06-29T18:12:45Z","snapshot_observed_at":"2026-08-09T04:44:37.252637Z","submitted_at":"2025-06-26T17:41:35Z","title":"Potemkin Understanding in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.21521","snapshot_observed_at":"2026-08-06T15:20:14.933677Z","title":"Potemkin understanding in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:14.933677Z"},"links":{"cited_paper":"/paper/2506.21521","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:0fed3fb07adaeeff809dcee60e450bf201c179f2d8d97378ef37cb2a2e3634c4","observation_id":"b920a734-a6e2-4c40-8dfb-4b2310026e8f","resolution":{"observed_at":"2026-08-06T15:20:14.933677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.987297Z","title":null,"venue":null,"work_id":"61b91cae-29aa-4d30-b121-2be97d4f16d6","year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.027056Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:d67ae5fa2b8c83159e2d0da756cb04bce07eec70e362e6b98536aaf3a150f395","observation_id":"6f496522-0762-4ba9-8b11-eb318a0481da","resolution":{"observed_at":"2026-08-06T15:20:19.991659Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.972504Z","title":null,"venue":null,"work_id":"9d8ea899-9d93-41f4-9c10-78d525299f90","year":1980},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.101001Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:23b971fb5a0f3f3eac07d2e731df71e91025252e6b43987d79f6b93fc3284ef5","observation_id":"ec17f54a-88df-4573-86ee-4ea54b1fe2e5","resolution":{"observed_at":"2026-08-06T15:20:19.977094Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13131","last_updated":"2025-06-16T06:37:18Z","snapshot_observed_at":"2026-08-07T04:21:43.190472Z","submitted_at":"2025-06-16T06:37:18Z","title":"AlphaEvolve: A coding agent for scientific and algorithmic discovery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13131","snapshot_observed_at":"2026-08-06T15:20:15.195563Z","title":"Z., Shirobokov, S., Kozlovskii, B., Ruiz, F","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.195563Z"},"links":{"cited_paper":"/paper/2506.13131","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:d74d1f4aa5d25dc98f1ab637688cd2b522e5114f067ad86dfc6ac22cdb41c24f","observation_id":"e03c38af-a9f6-442a-aa21-5a3b8c9380c6","resolution":{"observed_at":"2026-08-06T15:20:15.195563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:15.273173Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.273173Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:9003d7d3da35ff244161da3a8683bc8736d12a927fff7deae01d81f0bb9303cd","observation_id":"5279d530-2ec7-4e4b-ba4d-976e1581188c","resolution":{"observed_at":"2026-08-06T15:20:15.273173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14678","last_updated":"2025-02-20T16:09:55Z","snapshot_observed_at":"2026-08-07T18:01:35.596368Z","submitted_at":"2025-02-20T16:09:55Z","title":"How to Get Your LLM to Generate Challenging Problems for Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14678","snapshot_observed_at":"2026-08-06T15:20:15.359475Z","title":"How to get your llm to generate challenging problems for evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.359475Z"},"links":{"cited_paper":"/paper/2502.14678","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:da9bdf908df308c5d1ffdb69f2665fc7822f4d44881920096f6456bbcf16c83b","observation_id":"8bd3c511-51d1-44d7-a3b0-746a80f39589","resolution":{"observed_at":"2026-08-06T15:20:15.359475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.942059Z","title":"How does code pretraining affect language model task performance? Transactions on Machine Learning Research, 2025, 2025","venue":null,"work_id":"c532e0d4-7e39-4cf1-b47f-5037f7272d26","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.429571Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:5cac51ef526288ede4ec0323dae511a17752a84b8884c2771c18c4ada51c5b07","observation_id":"ba17a31f-ffb2-494e-9825-eb5d5ba96334","resolution":{"observed_at":"2026-08-06T15:20:19.946878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15143","last_updated":"2024-11-05T19:27:56Z","snapshot_observed_at":"2026-08-07T20:32:30.401426Z","submitted_at":"2024-11-05T19:27:56Z","title":"dafny-annotator: AI-Assisted Verification of Dafny Programs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15143","snapshot_observed_at":"2026-08-06T15:20:15.513785Z","title":"dafny-annotator: Ai-assisted verification of dafny programs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.513785Z"},"links":{"cited_paper":"/paper/2411.15143","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:427233c34ea0173cfd0c827adcc7d671d04064216e7f15c27339cf0c5f482c19","observation_id":"9e12d4bf-8609-4b5f-b343-b3ba71f9a748","resolution":{"observed_at":"2026-08-06T15:20:15.513785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.924169Z","title":"Qodo-Embed-1: State-of-the-Art Code Embedding Models","venue":null,"work_id":"f583037f-3b75-42b3-ab58-e405a839592a","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.587429Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:fd24cfc2040ee5de9c0467192d27bd1a6160ce0f4ef2ebc37708cdd06969ee53","observation_id":"86e6799f-2ae5-4ef2-87ce-33b062d9a77d","resolution":{"observed_at":"2026-08-06T15:20:19.928972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.906677Z","title":"What is ansible?, 2025","venue":null,"work_id":"b9e3fff1-a79d-46ae-acd9-66b520f67197","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.676924Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:2bf3e628194a7aba63f859da59b2dbe3e0c8f9a03c997c47fd5f1ad0f94f0426","observation_id":"e721aa21-6722-47e4-b27f-dc6af21dffa1","resolution":{"observed_at":"2026-08-06T15:20:19.912819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.889010Z","title":"Evaluating the ability of gpt-4o to generate verifiable specifications in verifast","venue":null,"work_id":"1a4ba2d6-5716-4aa7-ab73-3eba5c6aad9e","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.752691Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:c926500b6455b01def391fcb6bbef1e6cd46b3e626865527a6b136d83d01880e","observation_id":"cf835e8b-db0b-444a-b7a6-b6874a99dd82","resolution":{"observed_at":"2026-08-06T15:20:19.894470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.872662Z","title":"Quantifying contamination in evaluating code generation capabilities of language models","venue":null,"work_id":"19d73b93-b1cc-489e-9079-6ce3e7f4dd1d","year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.842211Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:b393ee07294331131aac0d1af9aee7cca1f8157ceb288c2876a8304edb7ca187","observation_id":"c8a69c7a-4f56-40da-a2a7-54fe9c1d5311","resolution":{"observed_at":"2026-08-06T15:20:19.878126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.856036Z","title":"R., Gnaneshwar, D., Locatelli, A., Kirk, R., Rockt \\\"a schel, T., Grefenstette, E., and Bartolo, M","venue":null,"work_id":"5ae9604b-9c05-402f-afe4-57dd6a4a674b","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.858731Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:d27a488ba709ddaf6dea6bde5f80b9cd59c86522e4e66d962390d2e7709f8a3a","observation_id":"2c5f3e9e-cc6b-42fd-b790-d480cd83c094","resolution":{"observed_at":"2026-08-06T15:20:19.861737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16905","last_updated":"2024-11-25T20:16:16Z","snapshot_observed_at":"2026-08-09T06:24:46.093821Z","submitted_at":"2024-11-25T20:16:16Z","title":"Boundless Socratic Learning with Language Games","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.16905","snapshot_observed_at":"2026-08-06T15:20:15.942232Z","title":"Boundless socratic learning with language games","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:15.942232Z"},"links":{"cited_paper":"/paper/2411.16905","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:efdc50d60af283f76ea1ced7e0825c13eef542623875efce37e929fa4abaa921","observation_id":"2258a791-bb9d-42a0-b019-fa488b484dcc","resolution":{"observed_at":"2026-08-06T15:20:15.942232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03170","last_updated":"2024-10-04T06:05:17Z","snapshot_observed_at":"2026-07-06T19:27:29.609371Z","submitted_at":"2024-10-04T06:05:17Z","title":"Autoregressive Large Language Models are Computationally Universal","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03170","snapshot_observed_at":"2026-08-06T15:20:16.110693Z","title":"Autoregressive large language models are computationally universal","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.110693Z"},"links":{"cited_paper":"/paper/2410.03170","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:747fa6d9fa790eea583e65a1496f60e35c9d781e1884df3b82dab046a4fa524a","observation_id":"b77dad33-2e0f-4bfe-a08c-b004e621c283","resolution":{"observed_at":"2026-08-06T15:20:16.110693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-06T15:20:16.225181Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.225181Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:ab4e412570225ff02c5ea90adfad261523b7f25a1ffc29f0ad2cbb88b798b489","observation_id":"c539490b-f77b-45b2-8265-0ddd3ebb8665","resolution":{"observed_at":"2026-08-06T15:20:16.225181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06941","last_updated":"2025-11-20T00:19:24Z","snapshot_observed_at":"2026-08-09T21:19:24.140229Z","submitted_at":"2025-06-07T22:42:29Z","title":"The Illusion of Thinking: Understanding the Strengths and Limitations of Reasoning Models via the Lens of Problem Complexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06941","snapshot_observed_at":"2026-08-06T15:20:16.310347Z","title":"The illusion of thinking: Understanding the strengths and limitations of reasoning models via the lens of problem complexity","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.310347Z"},"links":{"cited_paper":"/paper/2506.06941","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:5989229c9ded298e1b8a6c369be9ba20c73d26a857680841bd369f727500c52a","observation_id":"b4e2337f-c5da-4e47-9406-159cce9bb7c6","resolution":{"observed_at":"2026-08-06T15:20:16.310347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.837997Z","title":"and Sutton, R","venue":null,"work_id":"63cf1cfd-f414-4d07-9475-583d49c43f54","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.396651Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:1319590d220f6bd85d0cc46fbd3eb1ee0c600a4a3ac887a1080690b293f66b35","observation_id":"d73bd658-e949-4a95-955f-bfb56d68dad2","resolution":{"observed_at":"2026-08-06T15:20:19.844506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.820193Z","title":null,"venue":null,"work_id":"45cda197-fd5e-49fe-8c5b-259fb26818c2","year":2021},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.472998Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:296729eb97fe2f324d71863e34951afdf19b166b38f3d466cef3abc8c4e693b3","observation_id":"efb2e494-847d-4463-b65a-92102c56969a","resolution":{"observed_at":"2026-08-06T15:20:19.825059Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.801571Z","title":"Beyond semantics: The unreasonable effectiveness of reasonless intermediate tokens, May 2025","venue":null,"work_id":"0a7dc415-1506-40b6-89d1-ba850b436e22","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.595792Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:8f5aa289f139033377e4ad65e624d05b3061f4615d3fb27f8ecd9539658e0a6a","observation_id":"fe984bbe-0df2-48a7-8a76-f304ee47b47e","resolution":{"observed_at":"2026-08-06T15:20:19.805865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.785981Z","title":"Clover: Closed-loop verifiable code generation","venue":null,"work_id":"ccc464db-6ba2-4c3e-aeab-74fddaefa608","year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.778032Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:196a18c61c8f87356651c8dd8b7100e2979ef3d7b6213b93ea3f8788a92cfb9c","observation_id":"594c5252-7b8d-4089-9d1a-b3b203337122","resolution":{"observed_at":"2026-08-06T15:20:19.790933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.18880","last_updated":"2025-06-23T17:51:40Z","snapshot_observed_at":"2026-08-06T23:12:46.021492Z","submitted_at":"2025-06-23T17:51:40Z","title":"OMEGA: Can LLMs Reason Outside the Box in Math? Evaluating Exploratory, Compositional, and Transformative Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.18880","snapshot_observed_at":"2026-08-06T15:20:16.918871Z","title":"Omega: Can llms reason outside the box in math? evaluating exploratory, compositional, and transformative generalization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:16.918871Z"},"links":{"cited_paper":"/paper/2506.18880","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:0fcfe2a6d930efc2346b12fe724a73fbc3bf79ce551764031c8d5d9519e0dc0d","observation_id":"32cb581c-3213-4fb9-b9df-af707d223f04","resolution":{"observed_at":"2026-08-06T15:20:16.918871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.768111Z","title":"The bitter lesson","venue":null,"work_id":"c150da94-3c4a-48d1-a4f7-a19fc03aa925","year":2019},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.053344Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:26d827e71bf51030d75bdd067bdc110af21e12d31bff9c3765955a1a1e5a19c2","observation_id":"710d8701-93ce-4763-9efa-36aad7749b72","resolution":{"observed_at":"2026-08-06T15:20:19.773635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.749166Z","title":"S., Barto, A","venue":null,"work_id":"9b6c39d3-e293-47fb-8c33-f095c3f74ccd","year":1998},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.219865Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:cd896be3eaa8e92f871e7b06103657b617d64c073bebb0553ebf640c3c51ad7b","observation_id":"b1e6a56f-affc-46cd-b542-2f20a7c1d8e7","resolution":{"observed_at":"2026-08-06T15:20:19.755062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:17.337992Z","title":"S., McAllester, D., Singh, S., and Mansour, Y","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.337992Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:5846f66cd93b9cdda39d6796685ba6c7f188c2fc0b1b3e62fff466634899c468","observation_id":"644e4e30-d1c3-4bc3-a2dd-69b31d6bab0b","resolution":{"observed_at":"2026-08-06T15:20:17.337992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.717455Z","title":"K., Fu, S., and Sundaresan, N","venue":null,"work_id":"b6a576b2-80d9-4bd6-a377-354133f85170","year":2020},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.451410Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:bc215362f16f8b421536aeeba53401f251c33735a97d6ae0513f7877480dd206","observation_id":"2bb15353-3e51-4d5a-9a2f-abe251a4a5e0","resolution":{"observed_at":"2026-08-06T15:20:19.722750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.700262Z","title":"A promising path towards autoformalization and general artificial intelligence","venue":null,"work_id":"cffb918c-31f8-4eb8-8a0c-a47b7b78a565","year":2020},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.562609Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:39d1bc6dbd019c5045644b6c1999dd4e4a6c75f107bd1f931db766d02c0d3d7a","observation_id":"d07c9b94-f6be-4bb6-b3da-07a2646d8ea1","resolution":{"observed_at":"2026-08-06T15:20:19.705505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.681210Z","title":"Worldcoder, a model-based llm agent: Building world models by writing code and interacting with the environment","venue":null,"work_id":"cae005c8-bfa3-4d24-ad0c-c794c091bb53","year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.616061Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:31e411cee4665437473dde81603e1b3f8d3a118ffddfff948faa48ef7a926bbe","observation_id":"d086b80a-4169-474c-828d-c623390bb884","resolution":{"observed_at":"2026-08-06T15:20:19.686216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:17.647324Z","title":"Clever: A curated benchmark for formally verified code generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.647324Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:80adb3971ffd2fcbb12fec70a25a6f7b6c67a307763ccd4acdfe3f2e8b3b1bc6","observation_id":"503ceaa8-cfa8-4190-ba38-ea235016521b","resolution":{"observed_at":"2026-08-06T15:20:17.647324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.665843Z","title":null,"venue":null,"work_id":"dc9927cd-420c-40f8-b2fd-186dbcdad5e5","year":2021},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.652416Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:809590e82159158ea1a1996b668f9533488fcefac68cde4fe96e206f0d430ddc","observation_id":"55fda506-3489-4624-acac-f48830017611","resolution":{"observed_at":"2026-08-06T15:20:19.670359Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04197","last_updated":"2024-09-22T12:40:35Z","snapshot_observed_at":"2026-07-06T18:26:37.581248Z","submitted_at":"2024-06-06T15:55:53Z","title":"DICE: Detecting In-distribution Contamination in LLM's Fine-tuning Phase for Math Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04197","snapshot_observed_at":"2026-08-06T15:20:17.657163Z","title":"Dice: Detecting in-distribution contamination in llm's fine-tuning phase for math reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.657163Z"},"links":{"cited_paper":"/paper/2406.04197","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:d3f7d321c317021e3bce28d073749eae97893149298c304d1e97126a48cac02c","observation_id":"6f133273-40af-4c26-9b6c-a7c4f96aaf22","resolution":{"observed_at":"2026-08-06T15:20:17.657163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01231","last_updated":"2025-07-01T23:10:02Z","snapshot_observed_at":"2026-08-10T00:36:42.611689Z","submitted_at":"2025-07-01T23:10:02Z","title":"Rethinking the Illusion of Thinking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.01231","snapshot_observed_at":"2026-08-06T15:20:17.663386Z","title":"D., Romero-Sorozabal, P., Rocon, E., and Cebrian, M","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.663386Z"},"links":{"cited_paper":"/paper/2507.01231","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:a7220fed26199f502a34b96613bea4e200adeae6296e9ddc7ca3734107345141","observation_id":"135ff164-bace-489b-8e95-01e296fc3110","resolution":{"observed_at":"2026-08-06T15:20:17.663386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11354","last_updated":"2025-04-15T16:23:44Z","snapshot_observed_at":"2026-08-03T03:38:06.953412Z","submitted_at":"2025-04-15T16:23:44Z","title":"Kimina-Prover Preview: Towards Large Formal Reasoning Models with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.11354","snapshot_observed_at":"2026-08-06T15:20:17.669259Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.669259Z"},"links":{"cited_paper":"/paper/2504.11354","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:dccb713acc0ac1f3c991b8965c778cd3fabbe720d319f403b5d88e82ef1b6b11","observation_id":"3940a24d-4c04-407a-83cc-ab957e930820","resolution":{"observed_at":"2026-08-06T15:20:17.669259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:17.675213Z","title":"Reasoning or memorization? unreliable results of reinforcement learning due to data contamination","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.675213Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:3a651dd7ebb9051d76d5087018ea63823af376e06d8837a2c8379653916ea54a","observation_id":"6710087f-8224-4e35-ad50-9b82666406af","resolution":{"observed_at":"2026-08-06T15:20:17.675213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07266","last_updated":"2025-05-27T02:56:52Z","snapshot_observed_at":"2026-08-08T13:14:58.702776Z","submitted_at":"2025-02-11T05:28:59Z","title":"When More is Less: Understanding Chain-of-Thought Length in LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.07266","snapshot_observed_at":"2026-08-06T15:20:17.681074Z","title":"When more is less: Understanding chain-of-thought length in llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.681074Z"},"links":{"cited_paper":"/paper/2502.07266","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:c773b9ac4b91c717f5f39b9b9d338f5053249885b70312a441c0b5ef7594ce61","observation_id":"bb3a35eb-8f44-4213-afc6-37a9832981d2","resolution":{"observed_at":"2026-08-06T15:20:17.681074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14655","last_updated":"2025-04-20T15:28:16Z","snapshot_observed_at":"2026-08-10T10:10:21.478473Z","submitted_at":"2025-04-20T15:28:16Z","title":"LeetCodeDataset: A Temporal Dataset for Robust Evaluation and Efficient Training of Code LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14655","snapshot_observed_at":"2026-08-06T15:20:17.686438Z","title":"K., Sun, H., Wu, S., Hu, J., and Xu, X","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.686438Z"},"links":{"cited_paper":"/paper/2504.14655","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:a776ab57238fd175bde03ae9cb12f1fe3f0797a8b5e96700d8ff3821ed50fc3d","observation_id":"bd8c166a-4143-41dc-9d89-038eee7ac710","resolution":{"observed_at":"2026-08-06T15:20:17.686438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16075","last_updated":"2024-12-20T17:19:24Z","snapshot_observed_at":"2026-08-10T16:31:30.588988Z","submitted_at":"2024-12-20T17:19:24Z","title":"Formal Mathematical Reasoning: A New Frontier in AI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16075","snapshot_observed_at":"2026-08-06T15:20:17.692754Z","title":"Formal mathematical reasoning: A new frontier in ai","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.692754Z"},"links":{"cited_paper":"/paper/2412.16075","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:103a0704f073cf8cfc61a002ce141bbe948cf14103d9f9dcc09fca6f6e941ffd","observation_id":"0312daef-72cb-4a38-9b52-3a44af887e8b","resolution":{"observed_at":"2026-08-06T15:20:17.692754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:17.698237Z","title":"Verina: Benchmarking verifiable code generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.698237Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:2ea4dd30df881cecde950fb382af8a0b7246c939cbb90f04b81519ca95ac8ff8","observation_id":"02da9f95-122e-4711-96f7-034f9c0f2115","resolution":{"observed_at":"2026-08-06T15:20:17.698237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.646541Z","title":"FormalMATH : Benchmarking formal mathematical reasoning of large language models, May 2025","venue":null,"work_id":"531fbd7f-355d-4170-a136-0eeabe55159e","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.703588Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:5257826eca7d2ac45b3e813a6059f92ca92fe6870c5d946e227b1d5c70b38d62","observation_id":"430c51eb-9fed-4caf-a9db-e088615483e0","resolution":{"observed_at":"2026-08-06T15:20:19.652533Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.630360Z","title":"Does reinforcement learning really incentivize reasoning capacity in LLMs beyond the base model?, April 2025","venue":null,"work_id":"0475e57e-4589-4420-a13f-bef017fb9374","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.708408Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:443f6b1f835ec0c767cbd193dd2d25ed4dfe9fb8509a5296cf32ca0f1ccac35f","observation_id":"e8b21127-c75a-417a-a23d-bebf0320684d","resolution":{"observed_at":"2026-08-06T15:20:19.634829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03335","last_updated":"2025-10-16T08:23:36Z","snapshot_observed_at":"2026-07-06T21:19:34.329442Z","submitted_at":"2025-05-06T09:08:00Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.03335","snapshot_observed_at":"2026-08-06T15:20:17.714526Z","title":"Absolute zero: Reinforced self-play reasoning with zero data, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.714526Z"},"links":{"cited_paper":"/paper/2505.03335","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:2e3d587db6a9681f422eb220e3281553faf2003835a9854b41d4766fc922252c","observation_id":"3699dcec-3bec-4155-88fa-a731f34d77b2","resolution":{"observed_at":"2026-08-06T15:20:17.714526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.00110","last_updated":"2022-02-28T06:03:23Z","snapshot_observed_at":"2026-08-09T06:26:57.793642Z","submitted_at":"2021-08-31T23:21:12Z","title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.00110","snapshot_observed_at":"2026-08-06T15:20:17.720789Z","title":"M., and Polu, S","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.720789Z"},"links":{"cited_paper":"/paper/2109.00110","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:0c9c9980ff41df65f6969c641061fbe6638247e387a5ff9a33045e9709fb55f8","observation_id":"e4f32607-c403-4172-afa8-285d69505bee","resolution":{"observed_at":"2026-08-06T15:20:17.720789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08105","last_updated":"2025-04-07T21:05:50Z","snapshot_observed_at":"2026-07-06T19:31:18.802964Z","submitted_at":"2024-10-10T16:53:10Z","title":"What Makes Large Language Models Reason in (Multi-Turn) Code Generation?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08105","snapshot_observed_at":"2026-08-06T15:20:17.726515Z","title":"What makes large language models reason in (multi-turn) code generation?, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.726515Z"},"links":{"cited_paper":"/paper/2410.08105","citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:5c7d3316e7313980089e1f7fb2c4e7dcfde9d7c99ba3cba389d8bc114cb6a07e","observation_id":"5bd1a8ea-c079-4975-9f12-b3db1ff86f95","resolution":{"observed_at":"2026-08-06T15:20:17.726515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:19.614257Z","title":"Reasoning by superposition: A theoretical perspective on chain of continuous thought","venue":null,"work_id":"305fda11-19f5-4729-b449-abb2d7832290","year":2025},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.731672Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:1e1e91f3becfbdeb5b7508fc42ef2269bfd2a2634d3cb9ff5e5861bdbe868033","observation_id":"ce79385c-3a09-4593-93cb-ba2fe6fb31bd","resolution":{"observed_at":"2026-08-06T15:20:19.619677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:17.737922Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.737922Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:092feb847d237be6e10c66a21cbc0d2cee0e039dd77eac047c3c5d68aa01d78c","observation_id":"4afedf7c-1975-4939-aed2-58f01ada670f","resolution":{"observed_at":"2026-08-06T15:20:17.737922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:17.743149Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.743149Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:5d9f221226b889597117f55c2533247a21a6aaf4c09f288b28aac673cb738f78","observation_id":"14cf0dc2-8a5a-408b-a8c5-64e0b2998ac7","resolution":{"observed_at":"2026-08-06T15:20:17.743149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"submission/0000775","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:18.018812Z","title":"best exploration","venue":null,"work_id":"ecf51cd1-a51c-4e19-aa2f-454c27c6cc81","year":2010},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.749558Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:6fd0aff81095c36c1708f2b5768129cb6c5120cf6e0a3ae8405fd930873d934f","observation_id":"a4963fce-d01f-401b-b5a9-2f72c91700a3","resolution":{"observed_at":"2026-08-06T15:20:18.027458Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:20:17.762309Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny","version":4},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-06T15:20:17.762309Z"},"links":{"citing_paper":"/paper/2507.16331"},"observation_digest":"sha256:61ee9f40a2d91c8b95bf92e59a4d298f1710e2b732cd1f763f41836e0245b021","observation_id":"8e5d1c21-5536-4f7b-a76b-20ce97f17e2f","resolution":{"observed_at":"2026-08-06T15:20:17.762309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.16331","last_updated":"2026-07-03T09:17:03Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T04:15:41.914807Z","submitted_at":"2025-07-22T08:13:01Z","title":"Re:Form -- Reducing Human Annotations in Scalable Formal Software Verification with RL in LLMs: A Preliminary Study on Dafny"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":76,"verified_exact":3,"verified_fuzzy":19},"total_outbound_references":101},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 100 of 101 outbound references and 6 inbound Pith citation observations for arXiv:2507.16331."}