{"as_of":"2026-08-06T12:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b49921239a967074102528e8efc2f4ddc494600d065675395d1b5b20d6296e1e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":67,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":67,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":67,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":67,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T04:32:28.414316Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2505.12741","last_updated":"2026-05-31T07:24:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-19T05:56:06Z","title":"Language Model Networks: Supervision-Efficient Learning through Dense Communication","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-22T14:50:45.735917Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2505.12741"},"observation_digest":"sha256:58057f5bc72e9497775b71445d3b3b84ae4fa92d80ce25f751eac0d8bcbcb51a","observation_id":"6ff1a7ce-37e5-466c-9258-0f89657ce664","resolution":{"observed_at":"2026-05-22T14:51:41.922988Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2505.14362","last_updated":"2026-03-01T04:59:56Z","snapshot_observed_at":"2026-08-02T12:23:34.946873Z","submitted_at":"2025-05-20T13:48:11Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T14:42:56.565621Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2505.14362"},"observation_digest":"sha256:df2809324821e5445b6838d57cb7df4bf8208a016362da95ea0e2bf9c07a683f","observation_id":"08a67229-bf5c-4b5f-8c83-6836f7d9bf24","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":203,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:e7ee87872dbeb3656799411b5327d74a381daa9347b9374c16637f9ab30eaa2b","observation_id":"6ed33c05-2a02-4ce3-aa86-bc2c4b58f7ce","resolution":{"observed_at":"2026-05-14T22:23:15.079043Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-06T04:32:28.414316Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well? arXiv preprint arXiv:2503.24235 , 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.03444","last_updated":"2025-08-05T13:41:32Z","snapshot_observed_at":"2026-08-06T04:32:14.099496Z","submitted_at":"2025-08-05T13:41:32Z","title":"An Auditable Agent Platform For Automated Molecular Optimisation","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T04:32:28.414316Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2508.03444"},"observation_digest":"sha256:acc6b6f6173acdf996eeffeeefbe9c80905f93146f63d7106d014e92e1f7deb4","observation_id":"80f230ee-eecb-4e52-b148-cd867b81ab4a","resolution":{"observed_at":"2026-08-06T04:32:28.414316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2508.04204","last_updated":"2026-05-06T06:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-06T08:35:10Z","title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-19T01:02:07.088724Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2508.04204"},"observation_digest":"sha256:b336617fce59a088096bdbf70618f0ef73a5e4d1b5ff6dbf402990f2c9cb8559","observation_id":"5dd8f729-a7c6-45d6-8cbb-6940b1fb8c1c","resolution":{"observed_at":"2026-05-19T01:02:54.846516Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T22:05:41.175894Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.07598","last_updated":"2025-08-11T03:58:35Z","snapshot_observed_at":"2026-08-05T22:05:40.326216Z","submitted_at":"2025-08-11T03:58:35Z","title":"Keyword-Centric Prompting for One-Shot Event Detection with Self-Generated Rationale Enhancements","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-05T22:05:41.175894Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2508.07598"},"observation_digest":"sha256:fa363183c02e165a40f01def918c5daa94cc4819d1cc4b95d7248625ef53fe8c","observation_id":"0c4e2cd7-5941-4339-97ab-2d00cbb92056","resolution":{"observed_at":"2026-08-05T22:05:41.175894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2508.10164","last_updated":"2026-04-15T15:39:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-13T20:00:09Z","title":"Pruning Long Chain-of-Thought of Large Reasoning Models via Small-Scale Preference Optimization","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-18T22:26:52.748349Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2508.10164"},"observation_digest":"sha256:1add76fb3a5b41bf4a672e2d95cc22050a0a800129c6d1acc0e91f830c2d3597","observation_id":"aec27a67-80b0-4012-962a-548a8eece983","resolution":{"observed_at":"2026-05-18T22:31:53.158218Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T20:21:08.251549Z","title":"What, how, where, and how well? A survey on test-time scaling in large language models.CoRR, abs/2503.24235, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.10751","last_updated":"2025-08-14T15:34:47Z","snapshot_observed_at":"2026-08-05T20:21:02.654739Z","submitted_at":"2025-08-14T15:34:47Z","title":"Pass@k Training for Adaptively Balancing Exploration and Exploitation of Large Reasoning Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-05T20:21:08.251549Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2508.10751"},"observation_digest":"sha256:f4e73a8616b468c8a2c00228ef78dce4e7fbf933586ac8400778405b0b9eaf38","observation_id":"635baa46-0aa2-4c0c-83e7-d034e1aa98a7","resolution":{"observed_at":"2026-08-05T20:21:08.251549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T18:22:38.859325Z","title":"What, how, where, and how well? A survey on test-time scaling in large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.14703","last_updated":"2025-08-20T13:28:39Z","snapshot_observed_at":"2026-08-05T20:57:23.880109Z","submitted_at":"2025-08-20T13:28:39Z","title":"A Lightweight Incentive-Based Privacy-Preserving Smart Metering Protocol for Value-Added Services","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T18:22:38.859325Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2508.14703"},"observation_digest":"sha256:2d5aad897617906c8a0e4796ab4d12e44d23fca0165c6a86797a03cfbc2b0f06","observation_id":"54282c77-7e8e-4286-a164-1fef4fd1a4b1","resolution":{"observed_at":"2026-08-05T18:22:38.859325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T16:23:38.548888Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.18648","last_updated":"2025-08-27T02:51:03Z","snapshot_observed_at":"2026-08-05T16:23:31.235731Z","submitted_at":"2025-08-26T03:43:32Z","title":"Thinking Before You Speak: A Proactive Test-time Scaling Approach","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T16:23:38.548888Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2508.18648"},"observation_digest":"sha256:2fd7b8d2bde8cc812c6fd406208c6fb519484337eb59ba416d7b3cd19539ab2e","observation_id":"361b3c4b-d395-48b3-845d-fe628de702e8","resolution":{"observed_at":"2026-08-05T16:23:38.548888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2509.00084","last_updated":"2026-04-27T15:34:36Z","snapshot_observed_at":"2026-08-04T00:21:22.483235Z","submitted_at":"2025-08-27T06:51:48Z","title":"Learning to Refine: Self-Refinement of Parallel Reasoning in LLMs","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-18T21:16:15.703057Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.00084"},"observation_digest":"sha256:c27a203ed5912edcb5ceff2e736a17f4a8837bf38a9316bc478de8d00b1541c0","observation_id":"be4774e0-2ba7-44fc-8c21-b3f7d2849f9d","resolution":{"observed_at":"2026-05-18T21:16:51.034192Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T11:39:36.836416Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well?arXiv preprint arXiv:2503.24235, 2025b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.02350","last_updated":"2025-09-02T14:16:02Z","snapshot_observed_at":"2026-08-05T11:39:34.180867Z","submitted_at":"2025-09-02T14:16:02Z","title":"Implicit Reasoning in Large Language Models: A Comprehensive Survey","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-05T11:39:36.836416Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.02350"},"observation_digest":"sha256:90786af3e5605d29cd4d30f2a7a7491c16c231b108a432ff6e7c649d4d3f7919","observation_id":"bea2c57e-65ec-41ae-8802-3c9f90c5d6b3","resolution":{"observed_at":"2026-08-05T11:39:36.836416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T13:48:23.144764Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well? In arXiv:2503.24235,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-05T13:48:19.313917Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:23.144764Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:c28b0c4b7725e82901db093646ef0b1d21497dc779b726584fd9fc42ab59e77e","observation_id":"5ff7f1ea-4082-41c1-90fb-2770e16904bf","resolution":{"observed_at":"2026-08-05T13:48:23.144764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T05:33:35.588351Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well?, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.05209","last_updated":"2025-09-09T15:51:00Z","snapshot_observed_at":"2026-08-06T03:28:59.006409Z","submitted_at":"2025-09-05T16:11:05Z","title":"Hunyuan-MT Technical Report","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-05T05:33:35.588351Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.05209"},"observation_digest":"sha256:5c387e461d624543d3d5d029a3f858c6cecd1975225c53885686b5c8cdef01a2","observation_id":"1ca57269-32e3-4cf5-81e8-218e9eb3b865","resolution":{"observed_at":"2026-08-05T05:33:35.588351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-04T22:57:45.084185Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well? arXiv preprint arXiv:2503.24235, 2025 b","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.06923","last_updated":"2025-09-08T17:36:21Z","snapshot_observed_at":"2026-08-05T02:58:44.065696Z","submitted_at":"2025-09-08T17:36:21Z","title":"Staying in the Sweet Spot: Responsive Reasoning Evolution via Capability-Adaptive Hint Scaffolding","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-04T22:57:45.084185Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.06923"},"observation_digest":"sha256:8c07d8d64b0ad3b4a96b9a81a11f8cb4cf7befca75bbdc73258855acd0738e59","observation_id":"d7169412-fd5a-4bec-adf4-814bad7d7b24","resolution":{"observed_at":"2026-08-04T22:57:45.084185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-04T20:42:45.659388Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.08400","last_updated":"2025-09-10T08:44:57Z","snapshot_observed_at":"2026-08-04T20:42:44.961982Z","submitted_at":"2025-09-10T08:44:57Z","title":"Ubiquitous Intelligence Via Wireless Network-Driven LLMs Evolution","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T20:42:45.659388Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.08400"},"observation_digest":"sha256:ee16e78727b49f463c8db55046bbf3dc6430d9d4b7f150e9498954ea2fc03f60","observation_id":"06d80029-523d-43c1-af6e-a4aad1154b4a","resolution":{"observed_at":"2026-08-04T20:42:45.659388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2509.25758","last_updated":"2026-04-14T15:45:09Z","snapshot_observed_at":"2026-08-04T09:27:32.441799Z","submitted_at":"2025-09-30T04:23:43Z","title":"Thinking Sparks!: Emergent Attention Heads in Reasoning Models During Post Training","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-18T13:28:32.093512Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.25758"},"observation_digest":"sha256:1140d9c6e521619c19ba6ec1db5ccf30456a8f5a9c092d90d4558d5fe5b1f9a6","observation_id":"8ff00574-128a-4c4e-aa0f-46bfa8cce16f","resolution":{"observed_at":"2026-05-18T13:31:25.018073Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2510.20206","last_updated":"2026-05-14T08:53:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-23T04:45:09Z","title":"RAPO++: Cross-Stage Prompt Optimization for Text-to-Video Generation via Data Alignment and Test-Time Scaling","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-18T05:13:42.934115Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2510.20206"},"observation_digest":"sha256:f8d2ad06d5e07750655a6428913f92ab4ec1e5630c154e0db9a1ae4a5a27ebf7","observation_id":"92a13153-073c-4d94-bbe3-e1ab20c9601d","resolution":{"observed_at":"2026-05-18T05:15:54.301956Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2601.14249","last_updated":"2026-05-25T10:44:31Z","snapshot_observed_at":"2026-08-03T09:22:41.086646Z","submitted_at":"2026-01-20T18:58:10Z","title":"Which Reasoning Trajectories Teach Students to Reason Better? A Simple Metric of Informative Alignment","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T12:23:56.318846Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2601.14249"},"observation_digest":"sha256:73dc953de2ddd6ebe352cfa64ebc20f2bdcae8a6dee68fada94d15ac4d3410dd","observation_id":"24916366-268e-456d-9014-404a81e39914","resolution":{"observed_at":"2026-05-16T12:27:52.321859Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-03T09:22:48.116395Z","title":"What, how, where, and how well? A survey on test-time scaling in large language models.CoRR, abs/2503.24235, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.14249","last_updated":"2026-05-25T10:44:31Z","snapshot_observed_at":"2026-08-03T09:22:41.086646Z","submitted_at":"2026-01-20T18:58:10Z","title":"Which Reasoning Trajectories Teach Students to Reason Better? A Simple Metric of Informative Alignment","version":5},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-03T09:22:48.116395Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2601.14249"},"observation_digest":"sha256:4bd7b4fb6d556fbb846a56b6bebf487b6b928aebdd31114182e9e4ec9ff5f561","observation_id":"1d2d0a93-473d-447a-b361-53515bf69383","resolution":{"observed_at":"2026-08-03T09:22:48.116395Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2601.21484","last_updated":"2026-05-19T09:15:43Z","snapshot_observed_at":"2026-08-03T13:18:34.276013Z","submitted_at":"2026-01-29T10:06:52Z","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-16T09:37:57.120779Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2601.21484"},"observation_digest":"sha256:fbe8de2de533e34cbdd3875d5079080ca7d552b590e7702bcc76483c6d42bf77","observation_id":"74e38ed2-f0ba-45a3-b40e-388c51b635e6","resolution":{"observed_at":"2026-05-16T09:40:49.055978Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2601.21484","last_updated":"2026-05-19T09:15:43Z","snapshot_observed_at":"2026-08-03T13:18:34.276013Z","submitted_at":"2026-01-29T10:06:52Z","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-21T14:05:37.120262Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2601.21484"},"observation_digest":"sha256:6dc7f567e4e14a0e4e0541d0a642ac2fdad17b2be58909b5c7d8e3b05656206c","observation_id":"6d2fbad8-8e70-43d4-b9e2-5c2df885c689","resolution":{"observed_at":"2026-05-21T14:10:13.491193Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2603.08659","last_updated":"2026-04-06T09:23:11Z","snapshot_observed_at":"2026-08-03T00:34:34.031576Z","submitted_at":"2026-03-09T17:37:15Z","title":"CODA: Difficulty-Aware Compute Allocation for Adaptive Reasoning","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-15T14:23:00.793443Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2603.08659"},"observation_digest":"sha256:744e77b9603a69c2990e67c2e4ee3ad752114bb6e934c619b0c4a39de9f879bd","observation_id":"215eb138-c5f2-4e2a-91f3-5be6a610ed3e","resolution":{"observed_at":"2026-05-15T14:25:55.496930Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-07-14T22:25:04.080629Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well?arXiv preprint arXiv:2503.24235, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.12252","last_updated":"2026-06-18T07:56:55Z","snapshot_observed_at":"2026-07-14T22:25:02.474086Z","submitted_at":"2026-03-12T17:58:48Z","title":"EndoCoT: Scaling Endogenous Chain-of-Thought Reasoning in Diffusion Models","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-14T22:25:04.080629Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2603.12252"},"observation_digest":"sha256:496db452f16162186a3a333d90e1d6f2bf7a4c5de9524c1d0214f1837009fb05","observation_id":"5c9825f0-a7a8-4545-b0e5-e93100d8a2a9","resolution":{"observed_at":"2026-07-14T22:25:04.080629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-07-13T20:31:07.902275Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.22025","last_updated":"2026-06-27T01:25:31Z","snapshot_observed_at":"2026-08-02T05:11:13.037192Z","submitted_at":"2026-03-23T14:31:24Z","title":"Superbunched random fiber laser","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-07-13T20:31:07.902275Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2603.22025"},"observation_digest":"sha256:6f3b7a3cae2c5990811f915510903c6fb9db99bc93d57d67b0d7574bc066ab50","observation_id":"db38a336-7c6f-45b1-94e2-630e94abc879","resolution":{"observed_at":"2026-07-13T20:31:07.902275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.06262","last_updated":"2026-04-07T01:59:40Z","snapshot_observed_at":"2026-07-06T22:54:48.916799Z","submitted_at":"2026-04-07T01:59:40Z","title":"From Exposure to Internalization: Dual-Stream Calibration for In-context Clinical Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T19:19:15.053725Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.06262"},"observation_digest":"sha256:0f22edeee450d603298f97d00c93eeec5f704934c2b3cb79459af81623926390","observation_id":"29c3760e-7599-49a9-b847-138b70b74f46","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.10449","last_updated":"2026-04-12T04:15:31Z","snapshot_observed_at":"2026-07-06T22:59:05.519456Z","submitted_at":"2026-04-12T04:15:31Z","title":"AdverMCTS: Combating Pseudo-Correctness in Code Generation via Adversarial Monte Carlo Tree Search","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-10T16:35:16.056397Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.10449"},"observation_digest":"sha256:a87ff73c6092dc6f938a6e8105f9ccad39f9e498caec3c713028b825d18ee474","observation_id":"f2cf3053-f596-43c0-bcb9-400a852fa259","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.11025","last_updated":"2026-08-02T20:03:04Z","snapshot_observed_at":"2026-08-06T12:13:15.025422Z","submitted_at":"2026-04-13T05:49:04Z","title":"Test-time Scaling over Perception: Resolving the Grounding Paradox in Thinking with Images","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T16:38:11.785469Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.11025"},"observation_digest":"sha256:5a44ba3b412d14ce104e15d85c2dedd459d2d1ecf59149f4079b6719865a50a7","observation_id":"1202bc53-24f8-4823-97bf-cd004b5bcb83","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-04T05:31:43.668949Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.11025","last_updated":"2026-08-02T20:03:04Z","snapshot_observed_at":"2026-08-06T12:13:15.025422Z","submitted_at":"2026-04-13T05:49:04Z","title":"Test-time Scaling over Perception: Resolving the Grounding Paradox in Thinking with Images","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T05:31:43.668949Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.11025"},"observation_digest":"sha256:301f7249a0eb151ecbb1e286ee99d4424844814ff34c1f628af7f0b9a02e19bb","observation_id":"a05c182a-3b90-4dc6-be36-131d3b225e1a","resolution":{"observed_at":"2026-08-04T05:31:43.668949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.13552","last_updated":"2026-04-15T06:56:35Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:56:35Z","title":"Training-Free Test-Time Contrastive Learning for Large Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T14:02:10.277246Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.13552"},"observation_digest":"sha256:09f475902ce7b6579f91a5381b2c48032a1030639284fedb40db9e71fe026e23","observation_id":"45020902-06d3-4d75-8989-46652412f6b0","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.14564","last_updated":"2026-04-16T02:52:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-16T02:52:24Z","title":"MARS$^2$: Scaling Multi-Agent Tree Search via Reinforcement Learning for Code Generation","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T11:27:28.245835Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.14564"},"observation_digest":"sha256:62a38bc85b5bb647e58b0630c2a7e1be55698fbee2f2fc18fc0737cc6b4d19a3","observation_id":"9c331f8e-3d56-428e-99a8-1c05e5d0b725","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.14853","last_updated":"2026-04-16T10:39:22Z","snapshot_observed_at":"2026-07-06T23:02:31.717911Z","submitted_at":"2026-04-16T10:39:22Z","title":"Adaptive Test-Time Compute Allocation for Reasoning LLMs via Constrained Policy Optimization","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T11:57:44.680423Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.14853"},"observation_digest":"sha256:83f2fa5fe05804f3cbbbd4cbd4cfd146cddd1056d95a566fe5e5e9a16b09f15f","observation_id":"f85c44c7-9e02-4dc4-bacd-330e4c8661cb","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.17288","last_updated":"2026-04-19T07:04:49Z","snapshot_observed_at":"2026-07-06T23:04:23.730478Z","submitted_at":"2026-04-19T07:04:49Z","title":"Clover: A Neural-Symbolic Agentic Harness with Stochastic Tree-of-Thoughts for Verified RTL Repair","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T06:09:39.737812Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.17288"},"observation_digest":"sha256:89447163bca0af68a355ada67f631877fad0ed7c80ed585847b78f59acfa9379","observation_id":"f38c4e9e-bd5d-4679-a284-a33a0b9a2e63","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.17353","last_updated":"2026-04-19T09:59:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-19T09:59:35Z","title":"Hive: A Multi-Agent Infrastructure for Algorithm- and Task-Level Scaling","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T06:31:36.776819Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.17353"},"observation_digest":"sha256:e4cc0343db27d8a1c3d70016544336021419d8372cee7b0b69cf2b59e649a3ac","observation_id":"f8acaef1-d0f7-43c8-8a76-aa0792dea3ea","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.18356","last_updated":"2026-04-20T14:49:16Z","snapshot_observed_at":"2026-07-06T23:05:13.178333Z","submitted_at":"2026-04-20T14:49:16Z","title":"ComPASS: Towards Personalized Agentic Social Support via Tool-Augmented Companionship","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T04:54:54.317884Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.18356"},"observation_digest":"sha256:f282aa82043264bd37c430f11f328a63b4049cf794ed394342b7b089f23a0f5c","observation_id":"87813a6b-98f4-4a28-90b7-ed375d9be44f","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.19341","last_updated":"2026-04-21T11:24:09Z","snapshot_observed_at":"2026-07-06T23:06:02.635464Z","submitted_at":"2026-04-21T11:24:09Z","title":"Evaluation-driven Scaling for Scientific Discovery","version":1},"reference_index":172,"source":"pdf_text","source_observed_at":"2026-05-10T03:39:52.204043Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.19341"},"observation_digest":"sha256:4d6c3f152c04e4b995d83c507828d9d0a01304f6c20b2e24254bd9e72380ecd7","observation_id":"a8f18fe7-165c-4319-803b-cdbef7de8cda","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2604.26644","last_updated":"2026-08-03T06:41:06Z","snapshot_observed_at":"2026-08-06T12:17:19.356713Z","submitted_at":"2026-04-29T13:11:39Z","title":"When to Vote, When to Rewrite: Disagreement-Guided Strategy Routing for Test-Time Scaling","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-07T11:00:21.413246Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.26644"},"observation_digest":"sha256:c72356b27f60f12daeb86c6ef29874877b93ef4bd4f26d11d7aee38d00f86cd2","observation_id":"41daeef1-84a9-473a-9e10-044d5fbfcba4","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-04T05:23:31.417314Z","title":"Zhang, F","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.26644","last_updated":"2026-08-03T06:41:06Z","snapshot_observed_at":"2026-08-06T12:17:19.356713Z","submitted_at":"2026-04-29T13:11:39Z","title":"When to Vote, When to Rewrite: Disagreement-Guided Strategy Routing for Test-Time Scaling","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T05:23:31.417314Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2604.26644"},"observation_digest":"sha256:92c504252e3efbf18b581f965fcd19c9caf57d7fd43b2aacf2deca661d35dbe5","observation_id":"7c1b6035-7cef-45f8-ba83-0abc668be8af","resolution":{"observed_at":"2026-08-04T05:23:31.417314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.01194","last_updated":"2026-05-02T02:13:11Z","snapshot_observed_at":"2026-07-06T23:14:29.125042Z","submitted_at":"2026-05-02T02:13:11Z","title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-09T15:10:16.533927Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.01194"},"observation_digest":"sha256:1ef2d0bd65fd9924b3970c75de848463019dbb603230c6f8159f629b8621c7ce","observation_id":"040c9a7c-1c1a-4ddd-8a7a-fc16efeb42c6","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.04461","last_updated":"2026-05-06T03:40:05Z","snapshot_observed_at":"2026-08-01T01:47:15.158535Z","submitted_at":"2026-05-06T03:40:05Z","title":"Stream-T1: Test-Time Scaling for Streaming Video Generation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-08T18:16:14.985693Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.04461"},"observation_digest":"sha256:112adfa858ddc4ff548b7e34b369ad2e5515b62da7a7f1ba242ecaa0a2630517","observation_id":"cf2b85a0-724e-4107-90b1-9350996c6eb3","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.05561","last_updated":"2026-05-07T01:10:34Z","snapshot_observed_at":"2026-08-02T22:12:20.425304Z","submitted_at":"2026-05-07T01:10:34Z","title":"BitCal-TTS: Bit-Calibrated Test-Time Scaling for Quantized Reasoning Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-08T12:01:32.816855Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.05561"},"observation_digest":"sha256:62f998851f2bb2de06c0650254c29e47a82518e16f5d07582822d748180c553a","observation_id":"e3a48133-6027-4db9-a8af-e7fb77f0be0d","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.07177","last_updated":"2026-05-11T11:21:13Z","snapshot_observed_at":"2026-08-03T10:59:43.380213Z","submitted_at":"2026-05-08T03:16:08Z","title":"HyperEyes: Dual-Grained Efficiency-Aware Reinforcement Learning for Parallel Multimodal Search Agents","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-11T01:28:36.266167Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.07177"},"observation_digest":"sha256:a24d4a01bfbcd3fcbf0b5f16afe9748fb5be0f85acc205bcaf5db8e57a8c8284","observation_id":"168deb9b-cf9c-4a71-ae76-698f0a7ce50a","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.07177","last_updated":"2026-05-11T11:21:13Z","snapshot_observed_at":"2026-08-03T10:59:43.380213Z","submitted_at":"2026-05-08T03:16:08Z","title":"HyperEyes: Dual-Grained Efficiency-Aware Reinforcement Learning for Parallel Multimodal Search Agents","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-12T03:18:01.006274Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.07177"},"observation_digest":"sha256:0a439d47bc287b33e402d3ce1a4601c5748bebdb15497e7d61c7b06d7724197d","observation_id":"960fc558-7a55-4856-81c2-301c403ec0f8","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.08057","last_updated":"2026-05-08T17:44:15Z","snapshot_observed_at":"2026-07-31T05:49:37.295093Z","submitted_at":"2026-05-08T17:44:15Z","title":"CA-SQL: Complexity-Aware Inference Time Reasoning for Text-to-SQL via Exploration and Compute Budget Allocation","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-11T02:28:20.366674Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.08057"},"observation_digest":"sha256:a61b6d61c8586330c56f7863e43188f12ada04b2942c9bc0b17c1bb884dcfd65","observation_id":"695fbc2a-66c9-4b7c-b205-58776af2508d","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.08905","last_updated":"2026-05-09T11:57:25Z","snapshot_observed_at":"2026-07-06T23:21:06.773761Z","submitted_at":"2026-05-09T11:57:25Z","title":"Forge: Quality-Aware Reinforcement Learning for NP-Hard Optimization in LLMs","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-12T02:44:33.143247Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.08905"},"observation_digest":"sha256:dd4d66d322f9b97e2fbef68c7d98b0513cd56fe664f96490dbc5756ba6233c9e","observation_id":"b09f5760-0e30-4caf-a90f-3e35295e1a32","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.10754","last_updated":"2026-05-11T15:53:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-11T15:53:54Z","title":"The Agent Use of Agent Beings: Agent Cybernetics Is the Missing Science of Foundation Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-12T05:03:58.419364Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.10754"},"observation_digest":"sha256:1f1ede69ee94143d1f45b2f8b377a036fe205369e99c9e3a9a1be2443fee4a3c","observation_id":"e59d44d9-d712-45d1-b0f2-23acfeac2dd8","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.11625","last_updated":"2026-05-12T06:51:31Z","snapshot_observed_at":"2026-07-06T23:23:26.077921Z","submitted_at":"2026-05-12T06:51:31Z","title":"Nice Fold or Hero Call: Learning Budget-Efficient Thinking for Adaptive Reasoning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T01:31:21.377700Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.11625"},"observation_digest":"sha256:bb522112290cea29157d7fd190dd6c13679e75f3c702dca7228f3aa5a84d861a","observation_id":"1e831801-3de3-41b3-9d8a-9af4231f8f9e","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.11662","last_updated":"2026-05-12T07:22:42Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T07:22:42Z","title":"HSUGA: LLM-Enhanced Recommendation with Hierarchical Semantic Understanding and Group-Aware Alignment","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-13T01:06:31.734421Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.11662"},"observation_digest":"sha256:c73c59f5a87b0d272ab233d38913c5c50a1c69a79f4c63e7c79341ffc94e4f31","observation_id":"ae42288a-9be8-46e5-8e30-a62525f85f66","resolution":{"observed_at":"2026-05-13T18:07:52.994249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.18233","last_updated":"2026-05-18T11:28:45Z","snapshot_observed_at":"2026-08-01T23:39:03.834779Z","submitted_at":"2026-05-18T11:28:45Z","title":"Enhancing Train-Free Infinite-Frame Generation for Consistent Long Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T10:56:16.054496Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.18233"},"observation_digest":"sha256:52be05c5f97cfcff8e4f5094238578b9908038df6bb2f4973ae2373f62d35d6f","observation_id":"1b6aa264-76fe-4380-a872-7747055fc08c","resolution":{"observed_at":"2026-05-20T10:58:13.827505Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.25547","last_updated":"2026-05-25T08:03:31Z","snapshot_observed_at":"2026-08-03T06:30:00.604600Z","submitted_at":"2026-05-25T08:03:31Z","title":"TapSampling: Inference-Time Sampling with a Task-Progress-Understanding Verifier for Robotic Manipulation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-29T21:55:45.761276Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.25547"},"observation_digest":"sha256:b43ad084de217226999250fd4ccdbaa6311b480e2d26c21becaf8e882895e04f","observation_id":"8c48f4fd-a5ee-484b-abcf-b08f38131768","resolution":{"observed_at":"2026-06-29T22:04:00.515048Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.27030","last_updated":"2026-05-26T13:52:14Z","snapshot_observed_at":"2026-07-06T23:36:48.467513Z","submitted_at":"2026-05-26T13:52:14Z","title":"Share More, Search Less: Collaborative Parallel Thinking for Efficient Test-Time Scaling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T18:24:19.375500Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.27030"},"observation_digest":"sha256:4f853c07ae52ca348dd835249b14e8eb31d040c6e910bdd267cb36acd356a202","observation_id":"42aadb2a-f61b-467d-83a3-c564be5b5066","resolution":{"observed_at":"2026-06-29T18:33:51.112926Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.27596","last_updated":"2026-05-26T19:09:27Z","snapshot_observed_at":"2026-07-06T23:37:16.981482Z","submitted_at":"2026-05-26T19:09:27Z","title":"Can Hallucinations Be Useful? Solving Multi-Hop Questions With SLMs By Chaining System-I/II Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T18:27:56.586839Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.27596"},"observation_digest":"sha256:366a2735adc8f6537cbbd3b74820a91d58fa4fa7d8136ccbf751cd9ef86f1d6d","observation_id":"562a2218-671f-4753-b1d0-037a9886f0c1","resolution":{"observed_at":"2026-06-29T18:33:50.714947Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2605.31561","last_updated":"2026-05-29T17:27:07Z","snapshot_observed_at":"2026-08-03T06:37:03.004746Z","submitted_at":"2026-05-29T17:27:07Z","title":"What Am I Missing? Question-Answering as Hidden State Probing","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T22:17:50.790267Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2605.31561"},"observation_digest":"sha256:aef4e8484a4bcf454ed8390c5f5646254d7f2f17cd64191f93ba688b15f34b91","observation_id":"363cea04-bcd0-4043-8fd0-5b573bb6671c","resolution":{"observed_at":"2026-07-01T19:36:08.896349Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2606.03102","last_updated":"2026-06-02T03:42:04Z","snapshot_observed_at":"2026-08-03T04:21:57.609394Z","submitted_at":"2026-06-02T03:42:04Z","title":"Small RL Controller, Large Language Model: RL-Guided Adaptive Sampling for Test-Time Scaling","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-06-28T10:25:10.559953Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2606.03102"},"observation_digest":"sha256:1c84eaefcc64fd0b8018da4c961d33b92437de733252a0e32a5cef220cdfe237","observation_id":"4a9f7ac8-43c2-4712-adea-32e542c791e9","resolution":{"observed_at":"2026-07-02T02:56:30.014006Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2606.06906","last_updated":"2026-06-05T04:49:37Z","snapshot_observed_at":"2026-08-02T10:42:01.638389Z","submitted_at":"2026-06-05T04:49:37Z","title":"EASE-TTT: Evidence-Aligned Selective Test-Time Training for Long-Context Question Answering","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-06-27T22:05:00.537690Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2606.06906"},"observation_digest":"sha256:6a462aa7ea436d156ce3a0bb998088c714960812efe52ed3f2dd0eddaddaff64","observation_id":"03ae8aef-6ce3-4e79-83fb-803d3940b818","resolution":{"observed_at":"2026-07-02T17:17:15.117878Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2606.08231","last_updated":"2026-06-06T15:39:29Z","snapshot_observed_at":"2026-07-06T23:47:43.942733Z","submitted_at":"2026-06-06T15:39:29Z","title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","version":1},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-06-27T19:36:57.231932Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2606.08231"},"observation_digest":"sha256:9e055ebcf509e091a1254a289ec6cb28858a4462f929aa01eaa01f518a857612","observation_id":"7945aff2-afa2-4cab-b463-51f8cc0679e0","resolution":{"observed_at":"2026-07-02T21:37:25.411539Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-02T11:29:33.291784Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well?arXiv preprint arXiv:2503.24235, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.14502","last_updated":"2026-07-23T02:50:21Z","snapshot_observed_at":"2026-08-04T07:54:41.453095Z","submitted_at":"2026-06-12T14:33:55Z","title":"From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI","version":2},"reference_index":160,"source":"pdf_text","source_observed_at":"2026-08-02T11:29:33.291784Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2606.14502"},"observation_digest":"sha256:a684da2472cc5a9d8fb2a3c3f0ae317d1018486b92d682bcac9d2e71567aa277","observation_id":"cf5cc54c-97e8-46c7-b011-7809185b3168","resolution":{"observed_at":"2026-08-02T11:29:33.291784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2606.17890","last_updated":"2026-06-16T13:10:30Z","snapshot_observed_at":"2026-07-06T23:53:25.839293Z","submitted_at":"2026-06-16T13:10:30Z","title":"Dynamic Rollout Editing for Reducing Overthinking in RL-Trained Reasoning Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-06-27T01:19:55.164835Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2606.17890"},"observation_digest":"sha256:5c3814d6cb8fde94ea289ab147e4208df0f6dd7c072d79e8e43de602c22714bf","observation_id":"8eb7da15-fae3-4903-99e1-a21da45fda28","resolution":{"observed_at":"2026-06-27T01:20:20.424956Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2606.28661","last_updated":"2026-06-27T00:37:33Z","snapshot_observed_at":"2026-07-07T00:02:43.396508Z","submitted_at":"2026-06-27T00:37:33Z","title":"When More Sampling Hurts: The Modal Ceiling and Correlation Ceiling of Test-Time Scaling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-30T09:44:27.786630Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2606.28661"},"observation_digest":"sha256:b3001db5b13fa832b17da426735ab7deec732b39df89c0c6362483d7257890e9","observation_id":"588eb31e-9163-441f-aef9-3c09c37190df","resolution":{"observed_at":"2026-06-30T09:44:37.108721Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":"2503.24235","doi":"10.48550/arxiv.2503.24235","metadata_source":"pith","pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","venue":"cs.CL","work_id":"d4eaadf8-a3c6-4eee-98fb-1b337dd42e2d","year":2025},"citing_paper":{"arxiv_id":"2607.00399","last_updated":"2026-07-01T03:50:45Z","snapshot_observed_at":"2026-08-04T06:35:09.194548Z","submitted_at":"2026-07-01T03:50:45Z","title":"DriveVer: Lightweight Trajectory Evaluator as Test-Time Verifier for Autonomous Driving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-02T14:59:35.226245Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2607.00399"},"observation_digest":"sha256:b253fa884f0dc4dcba892e1c2113c15e07f52d68f0776d508f84baa871f5beab","observation_id":"f80e9360-d2ea-45da-8a55-ed340935577f","resolution":{"observed_at":"2026-07-02T15:07:04.062390Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-01T23:34:09.881076Z","title":"A survey on test- time scaling in large language models: What, how, where, and how well?arXiv preprint arXiv:2503.24235, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15388","last_updated":"2026-07-16T18:38:10Z","snapshot_observed_at":"2026-08-05T09:07:38.422534Z","submitted_at":"2026-07-16T18:38:10Z","title":"Precise but Uncoupled: Reviewer Precision Does Not Guarantee Critique Uptake in Multi-Agent Math Reasoning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T23:34:09.881076Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2607.15388"},"observation_digest":"sha256:b025dfb7f2b842c532bdaee088d9e7be80b678fb1cebfd02cfbf19af54fcc94f","observation_id":"c3f52520-79bc-49b9-9ab5-59ebf80e1509","resolution":{"observed_at":"2026-08-01T23:34:09.881076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-01T13:54:18.292672Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18979","last_updated":"2026-07-21T11:14:33Z","snapshot_observed_at":"2026-08-01T13:54:09.960101Z","submitted_at":"2026-07-21T11:14:33Z","title":"Fishing Out Free Riders: Shapley-Based Reward Attribution for Parallel Reasoning via Reinforcement Learning","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-01T13:54:18.292672Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2607.18979"},"observation_digest":"sha256:50231ab122c04cb2f0e84d7dc4f99d24377c4d370e63f34f973140559c899744","observation_id":"a6412470-d92e-45ab-876d-198c85f41806","resolution":{"observed_at":"2026-08-01T13:54:18.292672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-01T06:17:30.106561Z","title":"doi: 10.1137/0218082","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-03T19:51:32.725326Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":1989,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:30.106561Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:b8daec909a5c3332497e7060001b1c6fd6e0b12de46d526cb8651baf56309c48","observation_id":"f2543b24-09ef-4288-bffd-db2be8b8fe91","resolution":{"observed_at":"2026-08-01T06:17:30.106561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-01T01:25:07.947759Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.27788","last_updated":"2026-07-30T07:21:55Z","snapshot_observed_at":"2026-08-05T03:16:45.699632Z","submitted_at":"2026-07-30T07:21:55Z","title":"SpecCal: Ambiguity-Aware Candidate Calibration for Infrared Spectrum-Based Molecular Structure Reconstruction","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T01:25:07.947759Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2607.27788"},"observation_digest":"sha256:81d54d62639db43ce189168fb303bef3dabc7b94e93c8c6716af1c028271d014","observation_id":"b21b7678-ebc5-4a4b-b23a-6559ed780ed6","resolution":{"observed_at":"2026-08-01T01:25:07.947759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-06T00:24:22.524838Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well?arXiv preprint arXiv:2503.24235, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.01319","last_updated":"2026-08-02T15:41:12Z","snapshot_observed_at":"2026-08-06T12:15:26.483287Z","submitted_at":"2026-08-02T15:41:12Z","title":"Cognitive Demand Steering for Adaptive Meta-Reasoning in Large Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T00:24:22.524838Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2608.01319"},"observation_digest":"sha256:61fa00c08a09d1e0fc375163dbfe806741ef4f3f691c4971f6d60f5dfed9bdcb","observation_id":"5bff5347-115f-47a1-be8d-4f4b0d0e1253","resolution":{"observed_at":"2026-08-06T00:24:22.524838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T16:03:54.903882Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.03614","last_updated":"2026-08-04T13:04:42Z","snapshot_observed_at":"2026-08-06T11:39:38.018850Z","submitted_at":"2026-08-04T13:04:42Z","title":"Test-Time Scalable AI-RAN: Inference Time Allocation for Cell-Free MIMO","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T16:03:54.903882Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2608.03614"},"observation_digest":"sha256:cf62aa826305a3852b6eed4637c691d32642edb8ca895707708c76a5dfeef073","observation_id":"1f257e14-f413-474f-975e-6fdf0caf3909","resolution":{"observed_at":"2026-08-05T16:03:54.903882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T05:02:10.759585Z","title":"A survey on test-time scaling in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.03961","last_updated":"2026-08-04T17:27:21Z","snapshot_observed_at":"2026-08-06T11:41:03.976878Z","submitted_at":"2026-08-04T17:27:21Z","title":"Interpretable Adaptive Sampling for LLM Test-Time Scaling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T05:02:10.759585Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2608.03961"},"observation_digest":"sha256:64475db1f4c5206ea78d217ba326e05b23928d0ee04c3631e2d46b72c1a95d2c","observation_id":"1f76510a-7edb-4cfd-ba98-a124dbe8e1be","resolution":{"observed_at":"2026-08-05T05:02:10.759585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.24235/citation-record","integrity":"/paper/2503.24235/integrity","json":"/paper/2503.24235/citation-record.json","paper":"/paper/2503.24235"},"outbound":[],"paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-03T03:32:58.169782Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 67 inbound Pith citation observations for arXiv:2503.24235."}