{"as_of":"2026-08-07T19:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0a3b75b6bcf6a966dcaa31271fd498dc55b42a97734697a0cb93ed71fa3f61be","coverage":[{"denominator":217,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:07:39.908059Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-02T12:29:24.439779Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:36:56.135904Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"cited_work":{"arxiv_id":"2507.00004","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.00004","snapshot_observed_at":"2026-07-02T12:36:56.135904Z","title":"arXiv preprint arXiv:2507.00004 , year=","venue":null,"work_id":"8e757558-ceb0-44a5-a9e4-39ad19335615","year":null},"citing_paper":{"arxiv_id":"2607.00913","last_updated":"2026-07-01T13:18:21Z","snapshot_observed_at":"2026-08-05T09:20:57.270200Z","submitted_at":"2026-07-01T13:18:21Z","title":"Two AI Metrics Diverged: Will it Make All the Difference?","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-07-02T12:29:24.439779Z"},"links":{"cited_paper":"/paper/2507.00004","citing_paper":"/paper/2607.00913"},"observation_digest":"sha256:f6dfc614b768b9b7d98e90d80069d6ef7b2e7bdee37a5340bf9d6b60bfd76b00","observation_id":"5e5e584d-abf1-49fc-9722-19276522f94e","resolution":{"observed_at":"2026-07-02T12:36:56.137222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.00004/citation-record","integrity":"/paper/2507.00004/integrity","json":"/paper/2507.00004/citation-record.json","paper":"/paper/2507.00004"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-07T05:07:39.492720Z","title":"Scaling laws for neural language models,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.492720Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:e53d3fe00bd36e17005a734a47fceda367c29cf119f47d4115e9f9b3d26a6c42","observation_id":"ded0eeef-24b1-45e1-be13-4bc79ceb2429","resolution":{"observed_at":"2026-08-07T05:07:39.492720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.15556","last_updated":"2022-03-29T13:38:03Z","snapshot_observed_at":"2026-07-06T12:54:11.616335Z","submitted_at":"2022-03-29T13:38:03Z","title":"Training Compute-Optimal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.15556","snapshot_observed_at":"2026-08-07T05:07:39.498613Z","title":"Training compute-optimal large language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.498613Z"},"links":{"cited_paper":"/paper/2203.15556","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:c8378b0a6b98ca3b088dbc8d9446dead19a1d3fd0fdbb301204fbe63ada68a5e","observation_id":"bb086a52-8788-4114-a846-700cf8dbfb0e","resolution":{"observed_at":"2026-08-07T05:07:39.498613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15208","last_updated":"2025-04-21T16:26:56Z","snapshot_observed_at":"2026-08-07T16:00:32.194586Z","submitted_at":"2025-04-21T16:26:56Z","title":"Compute-Optimal LLMs Provably Generalize Better With Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15208","snapshot_observed_at":"2026-08-07T05:07:39.503440Z","title":"Compute-optimal LLMs provably generalize better with scale,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.503440Z"},"links":{"cited_paper":"/paper/2504.15208","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:bfa3d710248f6c67a8772132a2a849f94adcc70ba227f05014dd9a1db5e85e32","observation_id":"b0430e10-3c59-4534-997b-f8613924f1d5","resolution":{"observed_at":"2026-08-07T05:07:39.503440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.507915Z","title":"Training compute of frontier AI models grows by 4-5x per year,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.507915Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:814ba662fc66efc9b4fbd4877a3eca6291a5c99baf92f89eb16083fd2a525419","observation_id":"f0d827f6-3399-4524-9224-f3b73361d949","resolution":{"observed_at":"2026-08-07T05:07:39.507915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15377","last_updated":"2024-02-13T17:52:00Z","snapshot_observed_at":"2026-08-06T10:36:22.238109Z","submitted_at":"2023-11-26T18:36:28Z","title":"Increased Compute Efficiency and the Diffusion of AI Capabilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15377","snapshot_observed_at":"2026-08-07T05:07:39.512826Z","title":"Increased compute efficiency and the diffusion of AI capabilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.512826Z"},"links":{"cited_paper":"/paper/2311.15377","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:b6e745a37cb8699eacc73daf74a931484748a39d793cb26a10c386b56bc3ea7c","observation_id":"5efef214-44c5-4040-8b6f-e7af7d39138a","resolution":{"observed_at":"2026-08-07T05:07:39.512826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04305","last_updated":"2020-05-08T22:26:37Z","snapshot_observed_at":"2026-07-06T09:18:56.122938Z","submitted_at":"2020-05-08T22:26:37Z","title":"Measuring the Algorithmic Efficiency of Neural Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04305","snapshot_observed_at":"2026-08-07T05:07:39.517278Z","title":"Measuring the algorithmic efficiency of neural networks,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.517278Z"},"links":{"cited_paper":"/paper/2005.04305","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:02a7cb871dd44145c9d6482a2caefbbde759108caa01294dd40a89d89c0bedde","observation_id":"983dcfe7-b288-45fb-9d9d-5b300cfdfd01","resolution":{"observed_at":"2026-08-07T05:07:39.517278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05812","last_updated":"2024-03-09T06:26:21Z","snapshot_observed_at":"2026-07-06T17:41:55.780727Z","submitted_at":"2024-03-09T06:26:21Z","title":"Algorithmic progress in language models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05812","snapshot_observed_at":"2026-08-07T05:07:39.522279Z","title":"Algorithmic progress in language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.522279Z"},"links":{"cited_paper":"/paper/2403.05812","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:b3c785596efb29db964041aea0d95f6406600c1375d49bc9a268a67e38be654f","observation_id":"a40e8546-a6c2-4358-9f2c-4b0cc46c1a01","resolution":{"observed_at":"2026-08-07T05:07:39.522279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.526454Z","title":"Claude’s extended thinking,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.526454Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:478c85b568a2f7fa6eec771e415e89c8347d2d8b4ef29010ddcc0c33e46fb82f","observation_id":"55527bc7-82f6-45e1-a4e7-de716a995acb","resolution":{"observed_at":"2026-08-07T05:07:39.526454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T05:07:39.530848Z","title":"DeepSeek-R1: Incentivizing reasoning capability in LLMs via reinforcement learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.530848Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:588aae82b9432a567fc973ea3702834706a5850722bd21c38cb72c57a80e00c1","observation_id":"94aeaad4-b318-4e1c-8996-82fa3d566d7b","resolution":{"observed_at":"2026-08-07T05:07:39.530848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.534878Z","title":"Gemini 2.5: Our most intelligent AI model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.534878Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:e61a88ea7940bb1c5f947401322ca854e2d6b2965b90c362dc6d9605d9513641","observation_id":"d176cbbe-b1c1-4eeb-9c9f-b04f7c48f3cd","resolution":{"observed_at":"2026-08-07T05:07:39.534878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.538768Z","title":"IBM Granite 3.2: Reasoning, vision, forecasting and more,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.538768Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:8f50acb57d09ca2972bdc5c1c4224eec1dcdfaca7ec210f60b470fa8e04c8465","observation_id":"d7f1571f-5cf1-4269-b93e-c61995d18531","resolution":{"observed_at":"2026-08-07T05:07:39.538768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.21318","last_updated":"2025-04-30T05:05:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-30T05:05:09Z","title":"Phi-4-reasoning Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.21318","snapshot_observed_at":"2026-08-07T05:07:39.542544Z","title":"Phi-4- reasoning technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.542544Z"},"links":{"cited_paper":"/paper/2504.21318","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:1dc766f4ec60c8a2418fdc10c2b4b1bf2da99a200e5fbc459efce21d3c909ff7","observation_id":"be84ecfc-bee3-440e-87d9-358799f52e61","resolution":{"observed_at":"2026-08-07T05:07:39.542544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.546640Z","title":"OpenAI o1 System Card,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.546640Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:53e4b11c8dfcee5913873e40ce363969c9c62be7fbfa773fb6e7d1089c3e35f4","observation_id":"1fc145bf-976b-4c12-b63e-7551b5072a6c","resolution":{"observed_at":"2026-08-07T05:07:39.546640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.550360Z","title":"Introducing OpenAI o3 and o4-mini,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.550360Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:029816693cb8dfdbe90ea999cf424c6e9af0c533a7109da022bb10285012e4ba","observation_id":"0201b25d-8625-4501-8250-aa7f9357c89b","resolution":{"observed_at":"2026-08-07T05:07:39.550360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.554243Z","title":"Grok 3 Beta — The Age of Reasoning Agents,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.554243Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:8ee42becc0240e4f1d48a1fb327e8578f9ab87e1b4ada003607d230c9e07a8f1","observation_id":"b1190000-cdb1-46d9-90f8-e9f8ffd15ccb","resolution":{"observed_at":"2026-08-07T05:07:39.554243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.557955Z","title":"The growing energy footprint of artificial intelligence,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.557955Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:c9106d42d1998aa3878223403e7e9ce60248b8ea4c58c89f3f109e2174f267af","observation_id":"be22cfff-d3eb-4ee5-b11f-ad8d106bbf3b","resolution":{"observed_at":"2026-08-07T05:07:39.557955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.562176Z","title":"Estimating the carbon footprint of BLOOM, a 176B parameter language model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.562176Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:30897b13d6a1bb7f1be6247c83c323f0218eaef30b0a3535a5042860ced0ae25","observation_id":"77f11932-1a2d-439a-9e21-ecd23f414404","resolution":{"observed_at":"2026-08-07T05:07:39.562176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05149","last_updated":"2022-04-11T14:30:27Z","snapshot_observed_at":"2026-08-05T16:48:50.205106Z","submitted_at":"2022-04-11T14:30:27Z","title":"The Carbon Footprint of Machine Learning Training Will Plateau, Then Shrink","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05149","snapshot_observed_at":"2026-08-07T05:07:39.565848Z","title":"The carbon footprint of machine learning training will plateau, then shrink,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.565848Z"},"links":{"cited_paper":"/paper/2204.05149","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:6340d9f5b2f3ce6b300d00039ea7e04675c73690a37abc1688f149cb5e642f25","observation_id":"a8743a1b-2298-4a01-9984-c89d0c621898","resolution":{"observed_at":"2026-08-07T05:07:39.565848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.570414Z","title":"Sustainable AI: Environmental implications, challenges and opportunities,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.570414Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:38d9b6d6274b2f6bff64ba21d053906922b4fb003304e07207e77296a23f9366","observation_id":"d5477b62-9f63-4959-9c94-04557d63739a","resolution":{"observed_at":"2026-08-07T05:07:39.570414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.574308Z","title":"The next wave of AI: Demand and adoption,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.574308Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:6a406819b4ee3b44a82b361dd57a175d32d4d4666629644d9eef80383b8cb3d0","observation_id":"43c6e430-13ee-4da6-b7ea-0fefbd694ff9","resolution":{"observed_at":"2026-08-07T05:07:39.574308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16548","last_updated":"2025-06-13T15:59:59Z","snapshot_observed_at":"2026-07-06T20:27:04.664873Z","submitted_at":"2025-01-27T22:45:06Z","title":"From Efficiency Gains to Rebound Effects: The Problem of Jevons' Paradox in AI's Polarized Environmental Debate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16548","snapshot_observed_at":"2026-08-07T05:07:39.578208Z","title":"From efficiency gains to rebound effects: The problem of Jevons’ Paradox in AI’s polarized environmental debate,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.578208Z"},"links":{"cited_paper":"/paper/2501.16548","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:36d76b840a60a1247fdf4efeda7ced3431ca7832d4da8453b72d5592f796dc44","observation_id":"96a5ee41-d563-4a26-8ec6-1a7235947640","resolution":{"observed_at":"2026-08-07T05:07:39.578208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.582584Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.582584Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:154cf71807b2c11a68127e713e17f8e21c07d8eaf697d8d94ad050634e15b903","observation_id":"bd8f6d77-36a2-4910-a6a6-dcf1bfd64aae","resolution":{"observed_at":"2026-08-07T05:07:39.582584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.586504Z","title":"1B user messages sent on ChatGPT every day,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.586504Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:bcbf5a6d01114b9b2980c73a9137327ab3174584e1e761f9392060f87c5eb395","observation_id":"4cb6bb14-8ced-428d-8814-8dfb901fea59","resolution":{"observed_at":"2026-08-07T05:07:39.586504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.590585Z","title":"ChatGPT added one million users in the last hour,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.590585Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:32c401f68a8bc39b77e0c99e7dafe8dc8d4766423ca3a260adfe29edf473cf09","observation_id":"b3a8d728-3b7d-46d4-879a-679fc2647aee","resolution":{"observed_at":"2026-08-07T05:07:39.590585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.594451Z","title":"ChatGPT statistics and user trends (2025),","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.594451Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:1e6cb6c2ff236f64ba1b86f293d99c86a131a956bb52a711c1c811a28cba4e6f","observation_id":"97a61648-86eb-4065-9d7d-d6e0b30c8fbf","resolution":{"observed_at":"2026-08-07T05:07:39.594451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.598332Z","title":"A systematic review of Green AI,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.598332Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:eac453749eb7860c1fe62f02c1eb9c0c1691f9a60f65c226b505f1d8409b487e","observation_id":"c6daaac8-50e7-423f-b605-e0e4e0a7907d","resolution":{"observed_at":"2026-08-07T05:07:39.598332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.602174Z","title":"Deep Blue,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.602174Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:a3f38d705e225fe1ab3173ca9c5b745cd00963c73911510f95ef60640285e3a0","observation_id":"80a688e5-b70a-40ea-91c4-956cb869db40","resolution":{"observed_at":"2026-08-07T05:07:39.602174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.606172Z","title":"Mastering the game of Go with deep neural networks and tree search,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.606172Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:4914d24da478c123f007255c7ca1c43f9e875c0c5b2627f4c98cc7540b9c830a","observation_id":"6e7aebae-75b6-4a46-b0a6-9400c32231a5","resolution":{"observed_at":"2026-08-07T05:07:39.606172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.611183Z","title":"Mastering the game of Go without human knowledge,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.611183Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:7d77a81af9f230c0d936335da230ff03316d38ad5e80f02ccd2a57c56acb824a","observation_id":"e5c0c3f8-e05a-491c-b108-54348491f428","resolution":{"observed_at":"2026-08-07T05:07:39.611183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T05:07:39.615289Z","title":"Mastering Chess and Shogi by self-play with a general reinforcement learning algorithm,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.615289Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:32c107aa3cf03f79475595da6f482ffb1ab2ec52fbbd12ba08384c48ecef764b","observation_id":"3d9dc32f-db41-44a7-8c75-c83962b8fee6","resolution":{"observed_at":"2026-08-07T05:07:39.615289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03113","last_updated":"2021-04-15T10:03:37Z","snapshot_observed_at":"2026-07-06T10:57:14.668681Z","submitted_at":"2021-04-07T13:34:25Z","title":"Scaling Scaling Laws with Board Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.03113","snapshot_observed_at":"2026-08-07T05:07:39.619445Z","title":"Scaling scaling laws with board games,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.619445Z"},"links":{"cited_paper":"/paper/2104.03113","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:c686e1053dfa44779a86a1ce384e56274208ba53eac61bb5db6004228634e054","observation_id":"6f8b12d7-0413-43ee-87ec-b1f2fa7277c0","resolution":{"observed_at":"2026-08-07T05:07:39.619445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1705.02955","last_updated":"2017-11-16T23:12:15Z","snapshot_observed_at":"2026-07-06T05:41:45.343378Z","submitted_at":"2017-05-08T16:27:48Z","title":"Safe and Nested Subgame Solving for Imperfect-Information Games","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.02955","snapshot_observed_at":"2026-08-07T05:07:39.624213Z","title":"Safe and nested subgame solving for imperfect-information games,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.624213Z"},"links":{"cited_paper":"/paper/1705.02955","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:783d900b5bf2be1eb181cc2dc26b68988a54b9369a1dcf1c007a7f58553f6f3d","observation_id":"638d68f3-0448-4836-bc9a-b4bb22aa41b0","resolution":{"observed_at":"2026-08-07T05:07:39.624213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.628464Z","title":"Human-level play in the game of Diplomacy by combining language models with strategic reasoning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.628464Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:f536e3418bf4dcdf642bdcc56891e934d40139638e2df9c6d0e94a9561bd65a0","observation_id":"a549f08b-343c-4022-aca5-e6358296ec98","resolution":{"observed_at":"2026-08-07T05:07:39.628464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07682","last_updated":"2022-10-26T05:06:24Z","snapshot_observed_at":"2026-08-02T15:56:35.249569Z","submitted_at":"2022-06-15T17:32:01Z","title":"Emergent Abilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.07682","snapshot_observed_at":"2026-08-07T05:07:39.632468Z","title":"Emergent abilities of large language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.632468Z"},"links":{"cited_paper":"/paper/2206.07682","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:9b35b9a42dae53d249f7c6a87abdc3bc1ec90ddc5036f25a96b68bc7d32f9239","observation_id":"c111cbba-fc11-4128-a660-941e7adc9728","resolution":{"observed_at":"2026-08-07T05:07:39.632468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.636674Z","title":"An information theory of compute-optimal size scaling, emergence, and plateaus in language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.636674Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:817d27affdf156f671f42fae38336ea58262a62b08bc1448c0ee31e64c7224cb","observation_id":"f63715f7-46e5-4879-84ca-a789399dc89d","resolution":{"observed_at":"2026-08-07T05:07:39.636674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.640705Z","title":"Multi-task Language Understanding on MMLU Leaderboard,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.640705Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:5a174664179bda6dabf985fe63d918e224c0f69478f82e15b0c8c82e7bee0b01","observation_id":"d0251886-4a32-4390-8104-c926f87aa3af","resolution":{"observed_at":"2026-08-07T05:07:39.640705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.644826Z","title":"Measuring massive multitask language understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.644826Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:0a5e6a71290bb36e854df2623a2d41731e615cf815eb4544a23accf3d39c92b4","observation_id":"29c7dd85-4868-465f-8763-e1222da9e0d5","resolution":{"observed_at":"2026-08-07T05:07:39.644826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.648721Z","title":"Are emergent abilities of large language models a mirage?","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.648721Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:31580daa36fef30ff88862aa261dbe84686c6322299114787a4a9eab37646f40","observation_id":"2c45d034-17db-427c-a26d-ef2b36bfdb98","resolution":{"observed_at":"2026-08-07T05:07:39.648721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.652682Z","title":"The quantization model of neural scaling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.652682Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:bef6f00b0e1c5fcca16dc9bc04a4826614d6291bec170b2109cddbf7ed88b98b","observation_id":"e0d401d8-160a-41a9-9e0f-bb25ed697939","resolution":{"observed_at":"2026-08-07T05:07:39.652682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.656514Z","title":"Circuit tracing: Revealing computational graphs in language models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.656514Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:c33c8b19651505749f5f715c1364acef210828845b32a140533b6e9884e0d725","observation_id":"e6616e3c-20cf-4f50-8c31-bc0dc045550a","resolution":{"observed_at":"2026-08-07T05:07:39.656514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.664621Z","title":"Curriculum learning,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.664621Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:04f2ca9c22733f47b44addcc408746b2d8b139a77e52b4fb2e526ed53e55552e","observation_id":"3fa94251-d92d-4778-9dc5-ae14f5e8ac3f","resolution":{"observed_at":"2026-08-07T05:07:39.664621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15936","last_updated":"2023-11-06T00:36:24Z","snapshot_observed_at":"2026-07-06T16:00:08.643627Z","submitted_at":"2023-07-29T09:22:54Z","title":"A Theory for Emergence of Complex Skills in Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15936","snapshot_observed_at":"2026-08-07T05:07:39.668555Z","title":"A theory for emergence of complex skills in language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.668555Z"},"links":{"cited_paper":"/paper/2307.15936","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:41b88ba9007b8024ed28de133cb23fb3aee3bc1e17f78b3e6fea33845a4d565b","observation_id":"86c4006f-a309-4535-b9cf-08632f0485c9","resolution":{"observed_at":"2026-08-07T05:07:39.668555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.673016Z","title":"A mathematical theory for learning semantic languages by abstract learners,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.673016Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:e7ee4a2497bb3cfee170cdd635b2d6747e03df7e39d086b9b081810eddc54ee8","observation_id":"c12e0140-4c77-4313-8279-37d316de2be2","resolution":{"observed_at":"2026-08-07T05:07:39.673016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.676978Z","title":"Skill-Mix: a flexible and expandable family of evaluations for AI models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.676978Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:62ae8685403d5f453957fc2044534a6fac1f92aa44ddb4280cb3a85613a521a8","observation_id":"66468c21-ea12-4cdf-83f6-61997079e119","resolution":{"observed_at":"2026-08-07T05:07:39.676978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.680836Z","title":"The learning curve: implications of a quantitative analysis,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.680836Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:b7cc1102a249c940baa648749d975daaca044871375ea0a3ad6ab0090de73708","observation_id":"0cd33ccb-1cbf-4e55-ad77-ba6e7bff0b80","resolution":{"observed_at":"2026-08-07T05:07:39.680836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.684918Z","title":"Plateaus, dips, and leaps: Where to look for inventions and discoveries during skilled performance,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.684918Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:d31a806533df20891ada8c64b9e61e209d69610cf185e00642b21c1eb3396235","observation_id":"27c29121-cbce-4bce-83b4-97c2df0c216f","resolution":{"observed_at":"2026-08-07T05:07:39.684918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.688862Z","title":"A first-principles mathematical model integrates the disparate timescales of human learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.688862Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:6e4db3f4170271df2921a0a9604d8b322047b2a32944cce329840b442443936a","observation_id":"91c8e278-8ca1-4dfa-927a-b870f9af7cd7","resolution":{"observed_at":"2026-08-07T05:07:39.688862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.693324Z","title":"Spin-glass models as error-correcting codes,","venue":null,"work_id":null,"year":1989},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.693324Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:f061261ad6b0cf0d9f852435fe8289c2ecd20e271e9a677c58f396ed73a023d4","observation_id":"6bd50324-3c6f-4f18-971b-9b663a5a8c9f","resolution":{"observed_at":"2026-08-07T05:07:39.693324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.697193Z","title":"Newell,Unified Theories of Cognition","venue":null,"work_id":null,"year":1994},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.697193Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:c127252205dc179da345bddb995509a067c01e601cd0252c91f1ac1fac643b59","observation_id":"8491fe46-a0cd-40d9-9638-2b9ad7f199dd","resolution":{"observed_at":"2026-08-07T05:07:39.697193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.701454Z","title":"Barab ´asi,Network Science","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.701454Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:2b180a8f8afac6e7d9f7890757aed8629fb9ccc9ae406bde636eee5a342a4276","observation_id":"4c225486-1177-463a-bbeb-f745387c52fb","resolution":{"observed_at":"2026-08-07T05:07:39.701454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.705522Z","title":"Learning curves: Asymptotic values and rate of convergence,","venue":null,"work_id":null,"year":1993},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.705522Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:1165bd02bfec45a03d7061841c704ae9286372af06636df0e608bb7418bc95fe","observation_id":"b721141f-d0e6-4ec5-b9df-f670f1449ae9","resolution":{"observed_at":"2026-08-07T05:07:39.705522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.00409","last_updated":"2017-12-01T17:13:14Z","snapshot_observed_at":"2026-07-06T06:12:18.811024Z","submitted_at":"2017-12-01T17:13:14Z","title":"Deep Learning Scaling is Predictable, Empirically","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.00409","snapshot_observed_at":"2026-08-07T05:07:39.709672Z","title":"Deep learning scaling is predictable, empirically,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.709672Z"},"links":{"cited_paper":"/paper/1712.00409","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:65cdf273730e9c68523319194d5a99e63c37bceb773f95acf3aa97abebd7e856","observation_id":"27dac46d-595c-4af0-be03-16caa56f7f4b","resolution":{"observed_at":"2026-08-07T05:07:39.709672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.12673","last_updated":"2019-12-20T18:20:34Z","snapshot_observed_at":"2026-08-05T03:04:56.613105Z","submitted_at":"2019-09-27T13:27:53Z","title":"A Constructive Prediction of the Generalization Error Across Scales","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.12673","snapshot_observed_at":"2026-08-07T05:07:39.713752Z","title":"A constructive prediction of the generalization error across scales,","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.713752Z"},"links":{"cited_paper":"/paper/1909.12673","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:3e372e5ba3f28a92c2566d8ff597cc7067b11a4ed148a1d09b83f33d8f10f966","observation_id":"77e77b35-f651-4a00-96d2-619bda196821","resolution":{"observed_at":"2026-08-07T05:07:39.713752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-07T05:07:39.718254Z","title":"Language models are few-shot learners,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.718254Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:6da5b4cb1f5db1742d8b1589bc43f6a5e0ff6f1e2b492f1eee449c5dcedf2e87","observation_id":"b2483da6-1c56-4ecf-b3e0-8f16bcb95355","resolution":{"observed_at":"2026-08-07T05:07:39.718254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.722424Z","title":"Prediction and entropy of printed English,","venue":null,"work_id":null,"year":1951},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.722424Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:adc8ccdc10b19f7f66fbb359455ff8d7636c8c1f2c3854aa91bf8a99ad524fda","observation_id":"a5a4d548-51d1-4c27-b5a3-208835a5ce7a","resolution":{"observed_at":"2026-08-07T05:07:39.722424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.726859Z","title":"Explaining neural scaling laws,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.726859Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:fc0909b98466ed2cefde71054090f73d2e6bbe6d8c0c6b594975a099c0b4f5df","observation_id":"6366e3eb-1eff-489f-9840-95594bbd5bf6","resolution":{"observed_at":"2026-08-07T05:07:39.726859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.730817Z","title":"Towards a universal scaling law of LLM training and inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.730817Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:130a3b84fee10c7b2130f6de355b58aefc864560ceb574b9ddebb3bd316c7314","observation_id":"b54a65a2-cecf-44f0-8acc-597adfd2fe59","resolution":{"observed_at":"2026-08-07T05:07:39.730817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T05:07:39.734911Z","title":"The Llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.734911Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:c31b12278c847490f27a5035764d30b90b8eaaa880fd93a7b7316635a0a74da8","observation_id":"1e02fe25-2a3f-4352-8ea8-8dba24901087","resolution":{"observed_at":"2026-08-07T05:07:39.734911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04315","last_updated":"2024-12-06T11:39:27Z","snapshot_observed_at":"2026-08-04T05:54:36.205023Z","submitted_at":"2024-12-05T16:31:13Z","title":"Densing Law of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04315","snapshot_observed_at":"2026-08-07T05:07:39.739462Z","title":"Densing law of LLMs,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.739462Z"},"links":{"cited_paper":"/paper/2412.04315","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:d9a3da39bf7c9187d9b404052cdd781551233271981d322b162af3ea2a84a85c","observation_id":"7786d76e-84fb-4402-a578-d68742c0b14c","resolution":{"observed_at":"2026-08-07T05:07:39.739462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04757","last_updated":"2024-01-09T17:34:30Z","snapshot_observed_at":"2026-08-07T05:53:39.752355Z","submitted_at":"2024-01-09T17:34:30Z","title":"How predictable is language model benchmark performance?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04757","snapshot_observed_at":"2026-08-07T05:07:39.743953Z","title":"How predictable is language model benchmark performance?","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.743953Z"},"links":{"cited_paper":"/paper/2401.04757","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:d91a78537a1812a2c54b29911e1ca8f5ec712a48ecc1692ea22fb2f79346ba71","observation_id":"341e967e-f53c-44d0-b95a-640968e74a98","resolution":{"observed_at":"2026-08-07T05:07:39.743953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.10938","last_updated":"2024-10-01T23:38:10Z","snapshot_observed_at":"2026-07-06T18:15:54.402568Z","submitted_at":"2024-05-17T17:49:44Z","title":"Observational Scaling Laws and the Predictability of Language Model Performance","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.10938","snapshot_observed_at":"2026-08-07T05:07:39.748185Z","title":"Observational scaling laws and the predictability of language model performance,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.748185Z"},"links":{"cited_paper":"/paper/2405.10938","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:fa9b0df68c9c29a34f914b5f002dbd7fa065cb8a9e8f29357fe7109be4afd78a","observation_id":"55d90686-d264-4dcc-acd3-a88ee28203c9","resolution":{"observed_at":"2026-08-07T05:07:39.748185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08540","last_updated":"2024-06-14T20:21:05Z","snapshot_observed_at":"2026-07-06T17:43:56.860733Z","submitted_at":"2024-03-13T13:54:00Z","title":"Language models scale reliably with over-training and on downstream tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08540","snapshot_observed_at":"2026-08-07T05:07:39.752393Z","title":"Language models scale reliably with over-training and on downstream tasks,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.752393Z"},"links":{"cited_paper":"/paper/2403.08540","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:acd436bd797eada5a67f49d4fb904c69e215b3ef47d23d33547b664ab17cec43","observation_id":"7212905d-590a-4677-bc73-d486d3d50ec8","resolution":{"observed_at":"2026-08-07T05:07:39.752393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.14891","last_updated":"2023-07-24T00:05:04Z","snapshot_observed_at":"2026-08-07T16:10:09.620880Z","submitted_at":"2022-10-26T17:45:01Z","title":"Broken Neural Scaling Laws","version":17},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.14891","snapshot_observed_at":"2026-08-07T05:07:39.756787Z","title":"Broken neural scaling laws,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.756787Z"},"links":{"cited_paper":"/paper/2210.14891","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:176e8e49e5ce6e0bffc8c120fb34c953c22edc0f3ae3abefff9396cbc2141b46","observation_id":"f1e3447c-cb80-4840-93aa-5ed3ab6d4741","resolution":{"observed_at":"2026-08-07T05:07:39.756787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.761555Z","title":"Scaling laws for downstream task performance in machine translation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.761555Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:9bb10fbf6a434ee0e0e8d26416bd8de5cc6dc632c179cf8698025db7f4fe22ae","observation_id":"76fd424c-0c4b-41c9-8969-76a406233349","resolution":{"observed_at":"2026-08-07T05:07:39.761555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.02095","last_updated":"2021-10-05T14:49:00Z","snapshot_observed_at":"2026-07-06T11:54:34.626457Z","submitted_at":"2021-10-05T14:49:00Z","title":"Exploring the Limits of Large Scale Pre-training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.02095","snapshot_observed_at":"2026-08-07T05:07:39.765677Z","title":"Exploring the limits of large scale pre-training,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.765677Z"},"links":{"cited_paper":"/paper/2110.02095","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:643485f34c762442320234bf69a51411b813c42abee94505b49b306dc7996de5","observation_id":"24a0ee2a-be06-4e0c-ab17-dac540d80973","resolution":{"observed_at":"2026-08-07T05:07:39.765677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03201","last_updated":"2024-07-28T15:54:10Z","snapshot_observed_at":"2026-08-05T11:44:05.266284Z","submitted_at":"2023-07-05T15:32:21Z","title":"Scaling Laws Do Not Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.03201","snapshot_observed_at":"2026-08-07T05:07:39.769818Z","title":"Scaling laws do not scale,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.769818Z"},"links":{"cited_paper":"/paper/2307.03201","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:06c560abd51a180e5dcde6c4492eed32f8fbcc0444ae998fe9360322be87688c","observation_id":"0065568d-23ea-46ab-955f-c9b2bee4b5c9","resolution":{"observed_at":"2026-08-07T05:07:39.769818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.773993Z","title":"Not-just-scaling laws: Towards a better understanding of the downstream impact of language model design decisions,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.773993Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:82898f4ad2cb1b88caca7828948dd75db52d68f900b9b47fb969c68174cf091b","observation_id":"663ecb2d-e843-4e8e-8586-d528424257a7","resolution":{"observed_at":"2026-08-07T05:07:39.773993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.14199","last_updated":"2022-10-25T17:45:36Z","snapshot_observed_at":"2026-07-06T14:10:20.983444Z","submitted_at":"2022-10-25T17:45:36Z","title":"Same Pre-training Loss, Better Downstream: Implicit Bias Matters for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.14199","snapshot_observed_at":"2026-08-07T05:07:39.778229Z","title":"Same pre-training loss, better downstream: Implicit bias matters for language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.778229Z"},"links":{"cited_paper":"/paper/2210.14199","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:4c9c9c73336c905e513facb3e8d8a8b2fcd7aa5de416787d37fee35f8d27482c","observation_id":"798b2385-8575-4476-9590-01cbf7243f93","resolution":{"observed_at":"2026-08-07T05:07:39.778229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19206","last_updated":"2025-03-28T02:10:05Z","snapshot_observed_at":"2026-08-07T16:39:47.680195Z","submitted_at":"2025-03-24T23:11:56Z","title":"Overtrained Language Models Are Harder to Fine-Tune","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19206","snapshot_observed_at":"2026-08-07T05:07:39.782404Z","title":"Overtrained language models are harder to fine-tune,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.782404Z"},"links":{"cited_paper":"/paper/2503.19206","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:634281e94f0b21bb677615db1ffad04582aa7db9019716353b78899745754544","observation_id":"29406675-cb2b-4863-b3dd-2c3dd95f3e92","resolution":{"observed_at":"2026-08-07T05:07:39.782404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.786753Z","title":"Rethinking fine-tuning when scaling test-time compute: Limiting confidence improves mathematical reasoning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.786753Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:8378db00108a5eeb5c888e1e8e6fbd522741df606965665aeada5d3834beda15","observation_id":"4638aa3f-b743-452f-b280-1e9c4233a83c","resolution":{"observed_at":"2026-08-07T05:07:39.786753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.07958","last_updated":"2022-05-08T02:43:02Z","snapshot_observed_at":"2026-08-03T16:26:48.747700Z","submitted_at":"2021-09-08T17:15:27Z","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.07958","snapshot_observed_at":"2026-08-07T05:07:39.791105Z","title":"TruthfulQA: Measuring how models mimic human falsehoods,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.791105Z"},"links":{"cited_paper":"/paper/2109.07958","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:7d5c16efc81879f4db77915c50a26df2d191d2befe4ebe50f0a1de1a02f720db","observation_id":"e9dd4514-ab59-4454-848f-e22edde220ed","resolution":{"observed_at":"2026-08-07T05:07:39.791105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.08193","last_updated":"2022-03-16T01:35:45Z","snapshot_observed_at":"2026-07-06T11:58:21.596920Z","submitted_at":"2021-10-15T16:43:46Z","title":"BBQ: A Hand-Built Bias Benchmark for Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.08193","snapshot_observed_at":"2026-08-07T05:07:39.795184Z","title":"BBQ: A hand-built bias benchmark for question answering,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.795184Z"},"links":{"cited_paper":"/paper/2110.08193","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:89fb8e0cc44a5b22e37f02e1320e55c258a51fd85b8542255e8047a380d3a021","observation_id":"33d1bc75-12dc-4f7f-984d-dfbfd9f856a9","resolution":{"observed_at":"2026-08-07T05:07:39.795184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.02011","last_updated":"2023-05-24T06:55:50Z","snapshot_observed_at":"2026-08-07T09:09:12.525940Z","submitted_at":"2022-11-03T17:26:44Z","title":"Inverse scaling can become U-shaped","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.02011","snapshot_observed_at":"2026-08-07T05:07:39.799434Z","title":"Inverse scaling can become U-shaped,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.799434Z"},"links":{"cited_paper":"/paper/2211.02011","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:802cba2dd5d3b556ce2cd251339677a4acf79667606791e88909a437f814762c","observation_id":"50259960-870e-45ad-b591-0b717961a115","resolution":{"observed_at":"2026-08-07T05:07:39.799434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19393","last_updated":"2025-03-01T06:07:39Z","snapshot_observed_at":"2026-07-06T20:29:11.710285Z","submitted_at":"2025-01-31T18:48:08Z","title":"s1: Simple test-time scaling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19393","snapshot_observed_at":"2026-08-07T05:07:39.803725Z","title":"s1: Simple test-time scaling,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.803725Z"},"links":{"cited_paper":"/paper/2501.19393","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:24bcc59162a6170e79586b4c4ffaac5d3243da5e61d3c4a4871f9c8311cd5953","observation_id":"b52716c6-d685-48ae-a9a6-3271f62f6f1d","resolution":{"observed_at":"2026-08-07T05:07:39.803725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20571","snapshot_observed_at":"2026-08-07T05:07:39.807957Z","title":"Reinforcement learning for reasoning in large language models with one training example,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.807957Z"},"links":{"cited_paper":"/paper/2504.20571","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:8ea7719c32099cfa835a8372995199fda45e33da0865d667f88e6141f509cc40","observation_id":"5ad86b0f-93bd-4328-8700-4bce37b0ea1c","resolution":{"observed_at":"2026-08-07T05:07:39.807957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.11916","last_updated":"2023-01-29T05:14:17Z","snapshot_observed_at":"2026-08-07T04:02:08.445660Z","submitted_at":"2022-05-24T09:22:26Z","title":"Large Language Models are Zero-Shot Reasoners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.11916","snapshot_observed_at":"2026-08-07T05:07:39.812191Z","title":"Large language models are zero-shot reasoners,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.812191Z"},"links":{"cited_paper":"/paper/2205.11916","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:159787e8958819a5d1f2fdf62c5b7110872e700f0a0823f10d695ea6f1b17a5e","observation_id":"5cbeb825-46fa-49b6-809b-3a8ec04067d0","resolution":{"observed_at":"2026-08-07T05:07:39.812191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09261","last_updated":"2022-10-17T17:08:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-17T17:08:26Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.09261","snapshot_observed_at":"2026-08-07T05:07:39.815981Z","title":"Challenging BIG-Bench tasks and whether chain-of-thought can solve them,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.815981Z"},"links":{"cited_paper":"/paper/2210.09261","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:057f9063a9a95eb6ecb05584c23563fc867fcc1dd53dfe37427d95fd7781f420","observation_id":"2160b047-c53e-483c-8455-5b6271a68dcb","resolution":{"observed_at":"2026-08-07T05:07:39.815981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03268","last_updated":"2024-06-20T18:46:06Z","snapshot_observed_at":"2026-07-06T17:25:39.546901Z","submitted_at":"2024-02-05T18:25:51Z","title":"Understanding Reasoning Ability of Language Models From the Perspective of Reasoning Paths Aggregation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03268","snapshot_observed_at":"2026-08-07T05:07:39.820330Z","title":"Understanding reasoning ability of language models from the perspective of reasoning paths aggregation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.820330Z"},"links":{"cited_paper":"/paper/2402.03268","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:314a555946d886a4465c8a99ffc598b4222553482edd7bd24e9c579cf3c38429","observation_id":"6eb5c544-c2dc-4356-ae67-c3edea4480a2","resolution":{"observed_at":"2026-08-07T05:07:39.820330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.824523Z","title":"Sequence to sequence learning with neural networks: What a decade,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.824523Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:29a34f0258648e1986af69a558494b635a50d8a88670c5179cae7da6b89f839e","observation_id":"395dad7c-5bf2-4692-9418-3263c1e2e7c9","resolution":{"observed_at":"2026-08-07T05:07:39.824523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.828440Z","title":"AI doom from an LLM-plateau-ist perspective,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.828440Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:8f24499ac249b025e82d7d6e68505ee8a1a2f4f3827928c2fd2e763c86c4fc2f","observation_id":"43d03769-8801-4aad-8193-17701a25af25","resolution":{"observed_at":"2026-08-07T05:07:39.828440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.832462Z","title":"The first wave of AI innovation is over. here’s what comes next,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.832462Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:ae82111b0e43b966af2a7c2a483038a8a17123e0b6b0d5538552101b5220831d","observation_id":"dd5603ca-4f42-4b38-a44e-be59e9426aa4","resolution":{"observed_at":"2026-08-07T05:07:39.832462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.836408Z","title":"AI won’t plateau — if we give it time to think,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.836408Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:f3a275cffa0c182ccf4cfe038e753a46ab7e67b28110b70efe49517d42b25299","observation_id":"0a53bca0-df12-4823-90de-32112cda7561","resolution":{"observed_at":"2026-08-07T05:07:39.836408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.840145Z","title":"Scaling data-constrained language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.840145Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:6696daab7e4546ccdbe6277569b64adb17667c03e0da7a3c379d85458f79ed96","observation_id":"22f4293a-5b65-424e-ba50-474d1949f1d5","resolution":{"observed_at":"2026-08-07T05:07:39.840145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07177","last_updated":"2024-04-10T17:27:54Z","snapshot_observed_at":"2026-07-06T17:58:25.894552Z","submitted_at":"2024-04-10T17:27:54Z","title":"Scaling Laws for Data Filtering -- Data Curation cannot be Compute Agnostic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07177","snapshot_observed_at":"2026-08-07T05:07:39.844389Z","title":"Scaling laws for data filtering – data curation cannot be compute agnostic,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.844389Z"},"links":{"cited_paper":"/paper/2404.07177","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:80c28312bea1d78fa658abd9cdf6f2048a3ae0457d7c8fc65ff0630173a9ddd7","observation_id":"023ee840-7c37-4322-992c-200a388fecf4","resolution":{"observed_at":"2026-08-07T05:07:39.844389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21015","last_updated":"2025-02-07T23:27:07Z","snapshot_observed_at":"2026-08-06T12:15:04.691746Z","submitted_at":"2024-05-31T17:04:18Z","title":"The rising costs of training frontier AI models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21015","snapshot_observed_at":"2026-08-07T05:07:39.848494Z","title":"The rising costs of training frontier AI models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.848494Z"},"links":{"cited_paper":"/paper/2405.21015","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:764c88427e76de7c9e2f352bb3bc9d047fe57ebd9ddc92f3fd968380e64f7565","observation_id":"c6de41cc-bec8-40c6-bcd7-1aface91964a","resolution":{"observed_at":"2026-08-07T05:07:39.848494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.04388","last_updated":"2023-12-09T21:25:02Z","snapshot_observed_at":"2026-08-02T07:12:38.105035Z","submitted_at":"2023-05-07T22:44:25Z","title":"Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.04388","snapshot_observed_at":"2026-08-07T05:07:39.852434Z","title":"Language models don’t always say what they think: Unfaithful explanations in chain-of-thought prompting,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.852434Z"},"links":{"cited_paper":"/paper/2305.04388","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:e175834375447521d258f65fd327b391b400af1155b6ae16e241b86ebd317469","observation_id":"2fbe18fe-d46e-40fe-be04-734db104bfbf","resolution":{"observed_at":"2026-08-07T05:07:39.852434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01240","last_updated":"2023-03-02T03:54:28Z","snapshot_observed_at":"2026-07-06T13:59:15.998573Z","submitted_at":"2022-10-03T21:34:32Z","title":"Language Models Are Greedy Reasoners: A Systematic Formal Analysis of Chain-of-Thought","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01240","snapshot_observed_at":"2026-08-07T05:07:39.856507Z","title":"Language models are greedy reasoners: A systematic formal analysis of chain-of-thought,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.856507Z"},"links":{"cited_paper":"/paper/2210.01240","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:68ea489f4c7b86f4170d80f33266030e9117908f65f75fa29c6f82e1e71ca81c","observation_id":"8a60c330-fc02-4424-aa94-100361efb22b","resolution":{"observed_at":"2026-08-07T05:07:39.856507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17453","last_updated":"2024-04-07T00:56:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-29T17:59:56Z","title":"Efficient Streaming Language Models with Attention Sinks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17453","snapshot_observed_at":"2026-08-07T05:07:39.860710Z","title":"Efficient streaming language models with attention sinks,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.860710Z"},"links":{"cited_paper":"/paper/2309.17453","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:546676a2b92df47567cbc6b0fecba027a04353b1d3fdd2414062ae7b191a9c07","observation_id":"caa6192b-6ee1-41e8-a587-77f1a904d156","resolution":{"observed_at":"2026-08-07T05:07:39.860710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11154","last_updated":"2025-03-14T07:46:33Z","snapshot_observed_at":"2026-08-07T17:04:48.119124Z","submitted_at":"2025-03-14T07:46:33Z","title":"Don't Take Things Out of Context: Attention Intervention for Enhancing Chain-of-Thought Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.11154","snapshot_observed_at":"2026-08-07T05:07:39.864758Z","title":"Don’t take things out of context: Attention intervention for enhancing chain-of-thought reasoning in large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.864758Z"},"links":{"cited_paper":"/paper/2503.11154","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:91dd086ac7b0309530b3edfd536cdc63971212dce3a7287a60d8a7c5fdb25270","observation_id":"71ffee8b-c47a-4cfb-a28b-3ff54ef3c5ad","resolution":{"observed_at":"2026-08-07T05:07:39.864758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.868843Z","title":"The curse of CoT: On the limitations of chain-of-thought in in-context learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.868843Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:e12806fe68b683976c407955bc18c877ffcd8532470082438cff4c03936a7bca","observation_id":"489eb03e-4f65-440f-b1aa-4fd2ce6adefb","resolution":{"observed_at":"2026-08-07T05:07:39.868843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07266","last_updated":"2025-05-27T02:56:52Z","snapshot_observed_at":"2026-08-07T06:50:02.841285Z","submitted_at":"2025-02-11T05:28:59Z","title":"When More is Less: Understanding Chain-of-Thought Length in LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.07266","snapshot_observed_at":"2026-08-07T05:07:39.872688Z","title":"When more is less: Understanding chain-of-thought length in LLMs,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.872688Z"},"links":{"cited_paper":"/paper/2502.07266","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:b454c423cddb838de38194648c5aac35fe8285e05aae00780e068eb5781d33fe","observation_id":"634eb7db-c73e-4d8a-92e9-9d8456bc662d","resolution":{"observed_at":"2026-08-07T05:07:39.872688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-07T05:07:39.876992Z","title":"Let’s verify step by step,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.876992Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:515d88b665568d45aa24aeb4f3cf8024b0bf4f4716296d178111cec16d48525f","observation_id":"79cc6bdc-7dea-481c-bbe1-42664f1d2e42","resolution":{"observed_at":"2026-08-07T05:07:39.876992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T05:07:39.880977Z","title":"Training verifiers to solve math word problems,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.880977Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:f5d816d0d43c84bde4f0470268b2636a294834c5056e459de27c3820baa4cc9f","observation_id":"a884701b-03ed-4fe4-9a68-c650ccb5ee85","resolution":{"observed_at":"2026-08-07T05:07:39.880977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.884774Z","title":"When to solve, when to verify: Compute-optimal problem solving and generative verification for LLM reasoning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.884774Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:880ae690d5e381716782df91df564c8daedec8bdf49895d4e8c4d170fec9f5d6","observation_id":"e262f215-9b8c-4ac1-88db-db32dce362a4","resolution":{"observed_at":"2026-08-07T05:07:39.884774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-07T05:07:39.888499Z","title":"Self-consistency improves chain of thought reasoning in language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.888499Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:3008e27b4e33187a493490f61722579629091d0d99ea94b2bbcbf5a535c54eaf","observation_id":"f052279f-279d-41c7-83a1-2efabc1fa6e6","resolution":{"observed_at":"2026-08-07T05:07:39.888499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.892256Z","title":"Solving quantitative reasoning problems with language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.892256Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:b9ef8a1805a387b85b3edf737b917821dc6a15aac099b72db34bedf0eba5ffb6","observation_id":"bf2247ff-05b5-4065-9db4-a24a122bd5a1","resolution":{"observed_at":"2026-08-07T05:07:39.892256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17651","last_updated":"2023-05-25T19:13:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-30T18:30:01Z","title":"Self-Refine: Iterative Refinement with Self-Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17651","snapshot_observed_at":"2026-08-07T05:07:39.895889Z","title":"Self-Refine: Iterative refinement with self-feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.895889Z"},"links":{"cited_paper":"/paper/2303.17651","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:79e823e3b6738268473b9876d47af4b9f0b4ba4e5d203c018ae1e0a8ee50bca6","observation_id":"055e7f58-e5b6-429d-bf48-6546b15f8b3d","resolution":{"observed_at":"2026-08-07T05:07:39.895889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.899967Z","title":"Cost-of-Pass: An economic framework for evaluating language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.899967Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:d58468dd330ddfa47361fded9196922edaae9b53ab940a225790e61a1c46fa1c","observation_id":"bf4a5fc3-212a-4143-b14f-e9186f1fb833","resolution":{"observed_at":"2026-08-07T05:07:39.899967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2102.01293","last_updated":"2021-02-02T04:07:38Z","snapshot_observed_at":"2026-08-01T22:46:19.170916Z","submitted_at":"2021-02-02T04:07:38Z","title":"Scaling Laws for Transfer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.01293","snapshot_observed_at":"2026-08-07T05:07:39.903603Z","title":"Scaling laws for transfer,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.903603Z"},"links":{"cited_paper":"/paper/2102.01293","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:eaa1c1b038f566af2e608e91fee09063b4728d2713b51fe22ec8b242d900c41b","observation_id":"2aa08e6c-7e5a-4790-be4b-c704137801f3","resolution":{"observed_at":"2026-08-07T05:07:39.903603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:07:39.908059Z","title":"Reproducible scaling laws for contrastive language-image learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.908059Z"},"links":{"citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:0f5849af82ac7d5f213af044bbd2560b3e4aeb722af112412cad9c18395d0c05","observation_id":"45d542b2-0c07-4f17-ba0f-eb4f23d3ebc4","resolution":{"observed_at":"2026-08-07T05:07:39.908059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T04:57:52.653569Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":100,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":217},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 217 outbound references and 1 inbound Pith citation observation for arXiv:2507.00004."}