{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:W7JIC7LSBYMIHAPPPCCUIZ2H3G","short_pith_number":"pith:W7JIC7LS","schema_version":"1.0","canonical_sha256":"b7d2817d720e188381ef7885446747d9a05d26f1e52309d4b23b190a367fe105","source":{"kind":"arxiv","id":"2607.08646","version":1},"attestation_state":"computed","paper":{"title":"UltraX: Refining Pre-Training Data at Scale with Adaptive Programmatic Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dongsheng Liu, Hengyu Zhao, Jie Cai, Jie Zhou, Qiang Ma, XinLong Zhao, Xuanhe Zhou, Xu Han, Yudong Wang, Zheng Wang, Zhiyuan Liu, Zixuan Fu","submitted_at":"2026-07-09T16:18:07Z","abstract_excerpt":"As available training data approaches its physical limit, gains from Scaling Laws have begun to diminish. Consequently, improving Large Language Models (LLMs) now depends less on data expansion and more on higher-quality data utilization. However, in the context of large-scale corpora, existing refinement methodologies face significant limitations in quality, efficiency, and reliability: Rule-based approaches are constrained by fixed heuristics and struggle with instance-level variations; LLM-based approaches improve quality but fail to meet the efficiency and reliability requirements of large"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.08646","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-09T16:18:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cd473c08f1adc82ccfeddd392527f0e9f353dd173d7c1342d3236578f6b4582c","abstract_canon_sha256":"d34d5fc7dbf9beafd19f32f71033b40258905be5d8528096e0c85d4bde87c191"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-10T01:19:56.482029Z","signature_b64":"u6fUZPl33sqEdbdLWIKcHJ8rvKgF5VuUsnhiIR47yF5hwIhHogqXZB6OpjpJh4gpjmo4BkHV8eqwgBHnwxC0CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7d2817d720e188381ef7885446747d9a05d26f1e52309d4b23b190a367fe105","last_reissued_at":"2026-07-10T01:19:56.481612Z","signature_status":"signed_v1","first_computed_at":"2026-07-10T01:19:56.481612Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UltraX: Refining Pre-Training Data at Scale with Adaptive Programmatic Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dongsheng Liu, Hengyu Zhao, Jie Cai, Jie Zhou, Qiang Ma, XinLong Zhao, Xuanhe Zhou, Xu Han, Yudong Wang, Zheng Wang, Zhiyuan Liu, Zixuan Fu","submitted_at":"2026-07-09T16:18:07Z","abstract_excerpt":"As available training data approaches its physical limit, gains from Scaling Laws have begun to diminish. Consequently, improving Large Language Models (LLMs) now depends less on data expansion and more on higher-quality data utilization. However, in the context of large-scale corpora, existing refinement methodologies face significant limitations in quality, efficiency, and reliability: Rule-based approaches are constrained by fixed heuristics and struggle with instance-level variations; LLM-based approaches improve quality but fail to meet the efficiency and reliability requirements of large"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.08646","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.08646/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.08646","created_at":"2026-07-10T01:19:56.481664+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.08646v1","created_at":"2026-07-10T01:19:56.481664+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.08646","created_at":"2026-07-10T01:19:56.481664+00:00"},{"alias_kind":"pith_short_12","alias_value":"W7JIC7LSBYMI","created_at":"2026-07-10T01:19:56.481664+00:00"},{"alias_kind":"pith_short_16","alias_value":"W7JIC7LSBYMIHAPP","created_at":"2026-07-10T01:19:56.481664+00:00"},{"alias_kind":"pith_short_8","alias_value":"W7JIC7LS","created_at":"2026-07-10T01:19:56.481664+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G","json":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G.json","graph_json":"https://pith.science/api/pith-number/W7JIC7LSBYMIHAPPPCCUIZ2H3G/graph.json","events_json":"https://pith.science/api/pith-number/W7JIC7LSBYMIHAPPPCCUIZ2H3G/events.json","paper":"https://pith.science/paper/W7JIC7LS"},"agent_actions":{"view_html":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G","download_json":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G.json","view_paper":"https://pith.science/paper/W7JIC7LS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.08646&json=true","fetch_graph":"https://pith.science/api/pith-number/W7JIC7LSBYMIHAPPPCCUIZ2H3G/graph.json","fetch_events":"https://pith.science/api/pith-number/W7JIC7LSBYMIHAPPPCCUIZ2H3G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G/action/storage_attestation","attest_author":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G/action/author_attestation","sign_citation":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G/action/citation_signature","submit_replication":"https://pith.science/pith/W7JIC7LSBYMIHAPPPCCUIZ2H3G/action/replication_record"}},"created_at":"2026-07-10T01:19:56.481664+00:00","updated_at":"2026-07-10T01:19:56.481664+00:00"}