{"as_of":"2026-08-05T04:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4aeedf3c886eac55b56c6291bdaa32291b91430a9f2a6186cdedda0dd530dc2b","coverage":[{"denominator":14,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":14,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T09:08:13.233840Z","state":"measured"},{"denominator":14,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":14,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.22938/citation-record","integrity":"/paper/2606.22938/integrity","json":"/paper/2606.22938/citation-record.json","paper":"/paper/2606.22938"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":"2408.03314","doi":"10.18653/v1/2025.acl-long.1486","metadata_source":"pith","pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","venue":"cs.LG","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","year":2024},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:a5237e4b80a4548e364ef79678234eb2a1daf739e5f6ec4c0e79b86ff6f414b7","observation_id":"9fa85c59-7dd1-4ea6-b22c-6d09d0269ebf","resolution":{"observed_at":"2026-07-04T10:09:44.765381Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"cited_work":{"arxiv_id":"2504.20571","doi":"10.18653/v1/2024.acl-long.643","metadata_source":"pith","pith_arxiv_id":"2504.20571","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","venue":"cs.LG","work_id":"75a5258b-4143-4f2f-99f3-6d950a496305","year":2025},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"cited_paper":"/paper/2504.20571","citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:e31be08b7b4767c9d8fb44d211430ab00b3534db3dc66839f7aeec2df19badc2","observation_id":"92c70472-23b5-4c9f-a708-67a086483e8f","resolution":{"observed_at":"2026-07-04T10:09:44.762957Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:13e77e7c71e4a304c20f436ba05684a6360cd6c1b4a5a5270f27aee606c46bd4","observation_id":"459b3dad-a283-4b71-b0a6-6dfa33a41460","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":"That is, ˜g(s) :=E[τf] where τf := min{t≥0 :s 0 =s,head(s t) =f}","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:7e68a63db7e4a877fd584f8e7a61b9c942da048abefce0b612db986fe7f723f2","observation_id":"28e3b78d-e0c0-4aa0-9595-bf9717e659e9","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":"both states are absorbing)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:28a8a03f0db73d6cd55aca853167617cdbfa0ce5060e2b1f79deedfc28f1f98b","observation_id":"41ed3a94-9fc1-4355-af21-eb9ebaef0026","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":"In particular, this means that hx(s) =g(s) +q(s)H f","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:0a3a58a5a8e639a914e1f5660cbded5556dc00cfac52e5a7a5fa0efc6cd3b65c","observation_id":"71c59b06-81c8-4722-8296-47db34befad8","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":"In other words, µs is the expected number of visits to statesduring a target branch attempt","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:14fa085ce031eee6b915f55df761eb3672d172288cadf4887f5011ebc25f0bc6","observation_id":"a7fd09d5-7cad-41f6-bcb6-26b1627d7afd","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":"In other words, ˜µs is the expected number of visits to statesduring a non-target branch attempt until it goes back tof","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:b0898722ac2e9e20b0292616b6003057d55d767498fd6fa7ebc525792c640ad3","observation_id":"33be9614-87d8-48ce-8ed1-1a1d425efdac","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:e1fe7ed12db7bcf646e0a1b975bcc0bfc4fb3318dbb059486cb0fb95f284e3de","observation_id":"9f2de60b-c686-4cf8-a5d3-8a50a7a1ac21","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:7a17fa9582239843749a755e07ee539b2b4ab02307c3234ad02df97c2cab5291","observation_id":"06bfd5a6-4855-4f81-ba41-f64501295113","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:7786cb5723530f12b61fea7f3a281d0c03e89f1d1ba0ce6165fc35511e99f3ff","observation_id":"976dee28-bc8d-4e9c-bfcd-0bd1b94fec05","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:8389a2144f79303c3f995f7de44094a8154f6644bcf82f9af4f292880fcfa7d6","observation_id":"d5222742-6725-47e5-8d85-0c5c726d302f","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:bbe5f17f9ec2d5480a6854567eb3b513a54581118ffa7adfad363ee7af25f47d","observation_id":"19884a58-adca-48b8-b4a4-18a93de5ffd2","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T09:08:13.233840Z","title":"reached the state with headt i), letg i denote the expected time of first entry into R− K+1−i, and fi denote the expected time of first entry into L− K+1−i","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-26T09:08:13.233840Z"},"links":{"citing_paper":"/paper/2606.22938"},"observation_digest":"sha256:0634810f52da61cfa77566d71bbe1b61ed4455dd9759d8b6d957bdd04ae8a861","observation_id":"99a898e6-aab4-47cf-91ca-84a7d0c4a2a0","resolution":{"observed_at":"2026-06-26T09:08:13.233840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.22938","last_updated":"2026-06-22T07:16:08Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-03T14:16:39.308130Z","submitted_at":"2026-06-22T07:16:08Z","title":"Provable Benefits of RLVR over SFT for Reasoning Models: Learning to Backtrack Efficiently"},"reference_resolution":{"displayed":14,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":14},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 14 of 14 outbound references and 0 inbound Pith citation observations for arXiv:2606.22938."}