{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:LTY3KIMI745HVBQSO5UHKM37DC","short_pith_number":"pith:LTY3KIMI","canonical_record":{"source":{"id":"2604.13977","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-04-15T15:24:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e0ef6b2c5f370a70f5c90870ff151fe92aab8010ef24ee73d4ec5b622abc94b2","abstract_canon_sha256":"7c72be67364ce6a4314d5db0d044b0392f4ee56d9653b8b76b326287f8ab615f"},"schema_version":"1.0"},"canonical_sha256":"5cf1b52188ff3a7a8612776875337f18a4e2d936f5f9db01c9183f3bda6082f2","source":{"kind":"arxiv","id":"2604.13977","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.13977","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"arxiv_version","alias_value":"2604.13977v2","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.13977","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"pith_short_12","alias_value":"LTY3KIMI745H","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"pith_short_16","alias_value":"LTY3KIMI745HVBQS","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"pith_short_8","alias_value":"LTY3KIMI","created_at":"2026-07-31T01:33:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:LTY3KIMI745HVBQSO5UHKM37DC","target":"record","payload":{"canonical_record":{"source":{"id":"2604.13977","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-04-15T15:24:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e0ef6b2c5f370a70f5c90870ff151fe92aab8010ef24ee73d4ec5b622abc94b2","abstract_canon_sha256":"7c72be67364ce6a4314d5db0d044b0392f4ee56d9653b8b76b326287f8ab615f"},"schema_version":"1.0"},"canonical_sha256":"5cf1b52188ff3a7a8612776875337f18a4e2d936f5f9db01c9183f3bda6082f2","receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5cf1b52188ff3a7a8612776875337f18a4e2d936f5f9db01c9183f3bda6082f2","last_reissued_at":"2026-07-31T01:33:34.789940Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-31T01:33:34.789940Z"},"source_kind":"arxiv","source_id":"2604.13977","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T01:33:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"OYzbGmyVcA8cxMiiLuBAOQAUr3MITrTZl+QS9+8b2x/YSXL7M2Wg5aLamokDuz2vFvtj7HRXNsnc2LpqNbYzAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:13:50.453567Z"},"content_sha256":"852b2256098f32c2a599886e695efa8363933a9a327e643248f520664881a15f","schema_version":"1.0","event_id":"sha256:852b2256098f32c2a599886e695efa8363933a9a327e643248f520664881a15f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:LTY3KIMI745HVBQSO5UHKM37DC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"How Can We Synthesize High-Quality Pretraining Data? A Systematic Study of Prompt Design, Generator Model, and Source Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Rephrasing web text into structured formats like tables, FAQs, and math problems yields higher-quality synthetic pretraining data than raw web sources or prior synthetic techniques.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Atsuki Yamaguchi, Colin Raffel, Edward Emanuel Beeching, Elie Bakouch, Guilherme Penedo, Hynek Kydl\\'i\\v{c}ek, Joel Niklaus, Leandro Von Werra, Lewis Tunstall, Michal \\v{S}tef\\'anik, Thibaud Frere, Thomas Wolf","submitted_at":"2026-04-15T15:24:59Z","abstract_excerpt":"Synthetic data is a standard component in training large language models, yet systematic comparisons across design dimensions, including rephrasing strategy, generator model, and source data, remain absent. We conduct extensive controlled experiments, generating over one trillion tokens, to identify critical factors in rephrasing web text into synthetic pretraining data. Our results reveal that structured output formats, such as tables, math problems, FAQs, and tutorials, consistently outperform both curated web baselines and prior synthetic methods. Notably, increasing the size of the generat"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"structured output formats, such as tables, math problems, FAQs, and tutorials, consistently outperform both curated web baselines and prior synthetic methods. Notably, increasing the size of the generator model beyond 1B parameters provides no additional benefit. By applying our findings, we develop FinePhrase, a 486-billion-token open dataset of rephrased web text that outperforms all existing synthetic data baselines while reducing generation costs by up to 30 times.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That improvements measured in controlled experiments with smaller models and the chosen evaluation metrics will generalize to large-scale pretraining of frontier models and that the source data selection effects are not confounded by other training variables.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Rephrasing web text into structured formats such as tables, math problems, FAQs, and tutorials produces higher-quality synthetic pretraining data than curated web baselines or prior synthetic methods, as demonstrated by trillion-token experiments and the resulting FinePhrase dataset that reduces gen","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Rephrasing web text into structured formats like tables, FAQs, and math problems yields higher-quality synthetic pretraining data than raw web sources or prior synthetic techniques.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"822fd1ed2bc9732dbb02f16ad7a18bf0f59ef2ae4784f09f39461dd5f57451b2"},"source":{"id":"2604.13977","kind":"arxiv","version":2},"verdict":{"id":"8a2d9316-58b6-4bee-bd04-d6679fd19a58","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T13:21:42.591708Z","strongest_claim":"structured output formats, such as tables, math problems, FAQs, and tutorials, consistently outperform both curated web baselines and prior synthetic methods. Notably, increasing the size of the generator model beyond 1B parameters provides no additional benefit. By applying our findings, we develop FinePhrase, a 486-billion-token open dataset of rephrased web text that outperforms all existing synthetic data baselines while reducing generation costs by up to 30 times.","one_line_summary":"Rephrasing web text into structured formats such as tables, math problems, FAQs, and tutorials produces higher-quality synthetic pretraining data than curated web baselines or prior synthetic methods, as demonstrated by trillion-token experiments and the resulting FinePhrase dataset that reduces gen","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That improvements measured in controlled experiments with smaller models and the chosen evaluation metrics will generalize to large-scale pretraining of frontier models and that the source data selection effects are not confounded by other training variables.","pith_extraction_headline":"Rephrasing web text into structured formats like tables, FAQs, and math problems yields higher-quality synthetic pretraining data than raw web sources or prior synthetic techniques."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.13977/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"8a2d9316-58b6-4bee-bd04-d6679fd19a58"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T01:33:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IYvAMIQnjjO2tA/59x8/YpVaB1bvOiSq1aVOo7IPCBmjBJmcYVfnd69ZtvyLjrhzXcOrmIL3ZZkbNNL1FP2jCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:13:50.454726Z"},"content_sha256":"c5ae6b37be4f6b8ebaf835abe46fe420166c22b1f526fdf9fbd7d6e132b23196","schema_version":"1.0","event_id":"sha256:c5ae6b37be4f6b8ebaf835abe46fe420166c22b1f526fdf9fbd7d6e132b23196"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/LTY3KIMI745HVBQSO5UHKM37DC/bundle.json","state_url":"https://pith.science/pith/LTY3KIMI745HVBQSO5UHKM37DC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/LTY3KIMI745HVBQSO5UHKM37DC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T07:13:50Z","links":{"resolver":"https://pith.science/pith/LTY3KIMI745HVBQSO5UHKM37DC","bundle":"https://pith.science/pith/LTY3KIMI745HVBQSO5UHKM37DC/bundle.json","state":"https://pith.science/pith/LTY3KIMI745HVBQSO5UHKM37DC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/LTY3KIMI745HVBQSO5UHKM37DC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:LTY3KIMI745HVBQSO5UHKM37DC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"7c72be67364ce6a4314d5db0d044b0392f4ee56d9653b8b76b326287f8ab615f","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-04-15T15:24:59Z","title_canon_sha256":"e0ef6b2c5f370a70f5c90870ff151fe92aab8010ef24ee73d4ec5b622abc94b2"},"schema_version":"1.0","source":{"id":"2604.13977","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.13977","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"arxiv_version","alias_value":"2604.13977v2","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.13977","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"pith_short_12","alias_value":"LTY3KIMI745H","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"pith_short_16","alias_value":"LTY3KIMI745HVBQS","created_at":"2026-07-31T01:33:34Z"},{"alias_kind":"pith_short_8","alias_value":"LTY3KIMI","created_at":"2026-07-31T01:33:34Z"}],"graph_snapshots":[{"event_id":"sha256:c5ae6b37be4f6b8ebaf835abe46fe420166c22b1f526fdf9fbd7d6e132b23196","target":"graph","created_at":"2026-07-31T01:33:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"structured output formats, such as tables, math problems, FAQs, and tutorials, consistently outperform both curated web baselines and prior synthetic methods. Notably, increasing the size of the generator model beyond 1B parameters provides no additional benefit. By applying our findings, we develop FinePhrase, a 486-billion-token open dataset of rephrased web text that outperforms all existing synthetic data baselines while reducing generation costs by up to 30 times."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That improvements measured in controlled experiments with smaller models and the chosen evaluation metrics will generalize to large-scale pretraining of frontier models and that the source data selection effects are not confounded by other training variables."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Rephrasing web text into structured formats such as tables, math problems, FAQs, and tutorials produces higher-quality synthetic pretraining data than curated web baselines or prior synthetic methods, as demonstrated by trillion-token experiments and the resulting FinePhrase dataset that reduces gen"},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Rephrasing web text into structured formats like tables, FAQs, and math problems yields higher-quality synthetic pretraining data than raw web sources or prior synthetic techniques."}],"snapshot_sha256":"822fd1ed2bc9732dbb02f16ad7a18bf0f59ef2ae4784f09f39461dd5f57451b2"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2604.13977/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Synthetic data is a standard component in training large language models, yet systematic comparisons across design dimensions, including rephrasing strategy, generator model, and source data, remain absent. We conduct extensive controlled experiments, generating over one trillion tokens, to identify critical factors in rephrasing web text into synthetic pretraining data. Our results reveal that structured output formats, such as tables, math problems, FAQs, and tutorials, consistently outperform both curated web baselines and prior synthetic methods. Notably, increasing the size of the generat","authors_text":"Atsuki Yamaguchi, Colin Raffel, Edward Emanuel Beeching, Elie Bakouch, Guilherme Penedo, Hynek Kydl\\'i\\v{c}ek, Joel Niklaus, Leandro Von Werra, Lewis Tunstall, Michal \\v{S}tef\\'anik, Thibaud Frere, Thomas Wolf","cross_cats":["cs.AI","cs.LG"],"headline":"Rephrasing web text into structured formats like tables, FAQs, and math problems yields higher-quality synthetic pretraining data than raw web sources or prior synthetic techniques.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-04-15T15:24:59Z","title":"How Can We Synthesize High-Quality Pretraining Data? A Systematic Study of Prompt Design, Generator Model, and Source Data"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.13977","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-10T13:21:42.591708Z","id":"8a2d9316-58b6-4bee-bd04-d6679fd19a58","model_set":{"reader":"grok-4.3"},"one_line_summary":"Rephrasing web text into structured formats such as tables, math problems, FAQs, and tutorials produces higher-quality synthetic pretraining data than curated web baselines or prior synthetic methods, as demonstrated by trillion-token experiments and the resulting FinePhrase dataset that reduces gen","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Rephrasing web text into structured formats like tables, FAQs, and math problems yields higher-quality synthetic pretraining data than raw web sources or prior synthetic techniques.","strongest_claim":"structured output formats, such as tables, math problems, FAQs, and tutorials, consistently outperform both curated web baselines and prior synthetic methods. Notably, increasing the size of the generator model beyond 1B parameters provides no additional benefit. By applying our findings, we develop FinePhrase, a 486-billion-token open dataset of rephrased web text that outperforms all existing synthetic data baselines while reducing generation costs by up to 30 times.","weakest_assumption":"That improvements measured in controlled experiments with smaller models and the chosen evaluation metrics will generalize to large-scale pretraining of frontier models and that the source data selection effects are not confounded by other training variables."}},"verdict_id":"8a2d9316-58b6-4bee-bd04-d6679fd19a58"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:852b2256098f32c2a599886e695efa8363933a9a327e643248f520664881a15f","target":"record","created_at":"2026-07-31T01:33:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"7c72be67364ce6a4314d5db0d044b0392f4ee56d9653b8b76b326287f8ab615f","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-04-15T15:24:59Z","title_canon_sha256":"e0ef6b2c5f370a70f5c90870ff151fe92aab8010ef24ee73d4ec5b622abc94b2"},"schema_version":"1.0","source":{"id":"2604.13977","kind":"arxiv","version":2}},"canonical_sha256":"5cf1b52188ff3a7a8612776875337f18a4e2d936f5f9db01c9183f3bda6082f2","receipt":{"builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5cf1b52188ff3a7a8612776875337f18a4e2d936f5f9db01c9183f3bda6082f2","first_computed_at":"2026-07-31T01:33:34.789940Z","kind":"pith_receipt","last_reissued_at":"2026-07-31T01:33:34.789940Z","receipt_version":"0.3","signature_status":"unsigned_v0"},"source_id":"2604.13977","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:852b2256098f32c2a599886e695efa8363933a9a327e643248f520664881a15f","sha256:c5ae6b37be4f6b8ebaf835abe46fe420166c22b1f526fdf9fbd7d6e132b23196"],"state_sha256":"4d2a18176febfde879cbebc3e6c176d46ab988eecece2490b073fb9a2969def9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CcAhFrzmjX9uywDn/R/lyWqewy1/f8pqkeEq6eay6VvI4tXB9VFpxN7OCrELTrFmBqXgii8SPk0JblgYw0WZAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T07:13:50.460625Z","bundle_sha256":"c05f071b4b94cb0707710ac3dde1add0a77aa85f36d2949a0152646d6728b346"}}