{"as_of":"2026-08-10T23:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a3162a3c57ea1a795157ca4f0dc3f809abb70538c90cd0617d88d59b59d1ae05","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":8,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":8,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T23:33:58.134824Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T16:54:58.206077Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-08-03T23:33:58.134824Z","title":"Beyondweb: Lessons from scaling synthetic data for trillion-scale pretraining","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.05313","last_updated":"2026-07-13T06:02:36Z","snapshot_observed_at":"2026-08-07T04:09:13.749315Z","submitted_at":"2025-11-07T15:13:28Z","title":"Controllably Efficient Language Models","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-03T23:33:58.134824Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2511.05313"},"observation_digest":"sha256:545f62c05b57b671e620c45bb076eec9d9c8aefad106aee925b003d96a0a59d0","observation_id":"f0433471-6744-4cde-8e03-160a390f6dfa","resolution":{"observed_at":"2026-08-03T23:33:58.134824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":"2508.10975","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-06-30T16:54:58.206077Z","title":"BeyondWeb : Lessons from scaling synthetic data for trillion-scale pretraining","venue":null,"work_id":"353db54d-c3be-44a9-be23-922515a703bc","year":2022},"citing_paper":{"arxiv_id":"2511.23230","last_updated":"2026-04-04T17:59:32Z","snapshot_observed_at":"2026-08-02T21:48:00.220191Z","submitted_at":"2025-11-28T14:40:03Z","title":"Action-guided generation of 3D functionality segmentation data","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T04:49:06.269256Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2511.23230"},"observation_digest":"sha256:84a86aa26a8496b47d6cef425db20f8a2886df65e8d1c10f997d76f3f8cc75aa","observation_id":"62560294-e859-4955-9fd9-d16622566338","resolution":{"observed_at":"2026-05-17T04:51:31.948881Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-08-03T03:22:46.300889Z","title":"Beyondweb: Lessons from scaling synthetic data for trillion-scale pretraining","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.08275","last_updated":"2026-06-25T09:48:00Z","snapshot_observed_at":"2026-08-07T03:55:35.582600Z","submitted_at":"2026-02-09T05:10:26Z","title":"Linguistics and Human Brain: A Perspective of Computational Neuroscience","version":3},"reference_index":197,"source":"pdf_text","source_observed_at":"2026-08-03T03:22:46.300889Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2602.08275"},"observation_digest":"sha256:e95aadf50c33aebd29751b7f37042a8c967aefe849c5dd88a6382bc51b151464","observation_id":"4e028ef2-af96-4ef3-8b1b-581fe2349e5c","resolution":{"observed_at":"2026-08-03T03:22:46.300889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":"2508.10975","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-06-30T16:54:58.206077Z","title":"BeyondWeb : Lessons from scaling synthetic data for trillion-scale pretraining","venue":null,"work_id":"353db54d-c3be-44a9-be23-922515a703bc","year":2022},"citing_paper":{"arxiv_id":"2605.17775","last_updated":"2026-05-18T02:49:04Z","snapshot_observed_at":"2026-07-06T23:28:45.646975Z","submitted_at":"2026-05-18T02:49:04Z","title":"Systematic Evaluation of the Quality of Synthetic Clinical Notes Rephrased by LLMs at Million-Note Scale","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-20T11:44:58.613981Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2605.17775"},"observation_digest":"sha256:cb68031ac554c9a1a2337edb79a380d8ff9b5db6a6186739f1cd07e9e1035814","observation_id":"0f7db37f-fd7f-4ca6-ab9c-c886dea62ed6","resolution":{"observed_at":"2026-05-20T11:48:15.130856Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":"2508.10975","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-06-30T16:54:58.206077Z","title":"BeyondWeb : Lessons from scaling synthetic data for trillion-scale pretraining","venue":null,"work_id":"353db54d-c3be-44a9-be23-922515a703bc","year":2022},"citing_paper":{"arxiv_id":"2605.22769","last_updated":"2026-05-25T09:28:47Z","snapshot_observed_at":"2026-07-06T23:33:04.856017Z","submitted_at":"2026-05-21T17:31:17Z","title":"Understanding Data Temporality Impact on Large Language Models Pre-training","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T16:53:01.583415Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2605.22769"},"observation_digest":"sha256:cfdb6bd6f8f3ae852b590fdf9f333f5eb47757c6df026b77748b1b7ff22b78f8","observation_id":"b032f9a5-3ba8-490f-998e-2fa7cfccb265","resolution":{"observed_at":"2026-06-30T16:54:58.207894Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":"2508.10975","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-06-30T16:54:58.206077Z","title":"BeyondWeb : Lessons from scaling synthetic data for trillion-scale pretraining","venue":null,"work_id":"353db54d-c3be-44a9-be23-922515a703bc","year":2022},"citing_paper":{"arxiv_id":"2605.29548","last_updated":"2026-06-01T17:29:35Z","snapshot_observed_at":"2026-08-02T17:41:19.598860Z","submitted_at":"2026-05-28T08:02:11Z","title":"Why Larger Models Learn More: Effects of Capacity, Interference, and Rare-Task Retention","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-06-29T08:39:14.327838Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2605.29548"},"observation_digest":"sha256:09d903bbee067cedc2b852da4b088ccbfd04a408a53c135876c63e4ec100cf02","observation_id":"4f668e0a-0926-489d-8397-14e58271fe9e","resolution":{"observed_at":"2026-06-29T08:43:15.203279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-07-31T07:01:45.805431Z","title":"arXiv preprint arXiv:2508.10975 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24717","last_updated":"2026-07-27T17:54:12Z","snapshot_observed_at":"2026-08-08T00:53:22.078848Z","submitted_at":"2026-07-27T17:54:12Z","title":"DataOrchestra: Learning to Orchestrate Per-Example Curation of Pretraining Data","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-07-31T07:01:45.805431Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2607.24717"},"observation_digest":"sha256:29c1d5d2a2b9f53a047152147226900e1a1030589971d189b94326e33cf205ac","observation_id":"04a89781-b13e-4781-bf90-88f658abbf78","resolution":{"observed_at":"2026-07-31T07:01:45.805431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10975","snapshot_observed_at":"2026-08-01T03:02:07.696790Z","title":"doi:10.48550/arXiv.2508.10975 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25271","last_updated":"2026-07-28T04:18:49Z","snapshot_observed_at":"2026-08-10T14:39:39.683275Z","submitted_at":"2026-07-28T04:18:49Z","title":"Bridging Compute- and Data-Optimal Pretraining","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-08-01T03:02:07.696790Z"},"links":{"cited_paper":"/paper/2508.10975","citing_paper":"/paper/2607.25271"},"observation_digest":"sha256:089db7eced942017586b66f1fd1457d0bc4d785813eb2906ae3dfe2bdf9132bc","observation_id":"afc736d1-078d-4fbd-a7b6-8bed5d1cf7f6","resolution":{"observed_at":"2026-08-01T03:02:07.696790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.10975/citation-record","integrity":"/paper/2508.10975/integrity","json":"/paper/2508.10975/citation-record.json","paper":"/paper/2508.10975"},"outbound":[],"paper":{"arxiv_id":"2508.10975","last_updated":"2025-08-19T17:56:36Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-05T20:14:57.566953Z","submitted_at":"2025-08-14T17:55:47Z","title":"BeyondWeb: Lessons from Scaling Synthetic Data for Trillion-scale Pretraining"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 8 inbound Pith citation observations for arXiv:2508.10975."}