{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L4U62CEJ3MZHNHR6QVT3XHN3Q5","short_pith_number":"pith:L4U62CEJ","schema_version":"1.0","canonical_sha256":"5f29ed0889db32769e3e8567bb9dbb87697e8489663639864cb1dcf58de67fbd","source":{"kind":"arxiv","id":"2507.03253","version":2},"attestation_state":"computed","paper":{"title":"RefineX: Learning to Refine Pre-training Data at Scale from Expert-Guided Programs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baolong Bi, Dayiheng Liu, Jiafeng Guo, Junfeng Fang, Junyang Lin, Lingrui Mei, Shenghua Liu, Xingzhang Ren, Xueqi Cheng, Yiwei Wang","submitted_at":"2025-07-04T02:19:58Z","abstract_excerpt":"The foundational capabilities of large language models (LLMs) are deeply influenced by the quality of their pre-training corpora. However, enhancing data quality at scale remains a significant challenge, primarily due to the trade-off between refinement effectiveness and processing efficiency. While rule-based filtering remains the dominant paradigm, it typically operates at the document level and lacks the granularity needed to refine specific content within documents. Inspired by emerging work such as ProX, we propose $\\textbf{RefineX}$, a novel framework for large-scale, surgical refinement"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.03253","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-04T02:19:58Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6ee1312a6fba31d7764c615764d0be9eb045747bd107ed3ff3482252327d438e","abstract_canon_sha256":"901fa49ef441e21362877f45a48de181525a9d4db53e7833fc89c1ab105784f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:11.260436Z","signature_b64":"5L0h2Hrt+g0xoCxb3dIO6kHk3XbssZ8ZWh5o53b2rSaQ92BL2XJkfSC3v7rU+uDIoMo/exhCo+WmX0GrjRIXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f29ed0889db32769e3e8567bb9dbb87697e8489663639864cb1dcf58de67fbd","last_reissued_at":"2026-07-05T11:34:11.259990Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:11.259990Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RefineX: Learning to Refine Pre-training Data at Scale from Expert-Guided Programs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baolong Bi, Dayiheng Liu, Jiafeng Guo, Junfeng Fang, Junyang Lin, Lingrui Mei, Shenghua Liu, Xingzhang Ren, Xueqi Cheng, Yiwei Wang","submitted_at":"2025-07-04T02:19:58Z","abstract_excerpt":"The foundational capabilities of large language models (LLMs) are deeply influenced by the quality of their pre-training corpora. However, enhancing data quality at scale remains a significant challenge, primarily due to the trade-off between refinement effectiveness and processing efficiency. While rule-based filtering remains the dominant paradigm, it typically operates at the document level and lacks the granularity needed to refine specific content within documents. Inspired by emerging work such as ProX, we propose $\\textbf{RefineX}$, a novel framework for large-scale, surgical refinement"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.03253","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.03253/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.03253","created_at":"2026-07-05T11:34:11.260046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.03253v2","created_at":"2026-07-05T11:34:11.260046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.03253","created_at":"2026-07-05T11:34:11.260046+00:00"},{"alias_kind":"pith_short_12","alias_value":"L4U62CEJ3MZH","created_at":"2026-07-05T11:34:11.260046+00:00"},{"alias_kind":"pith_short_16","alias_value":"L4U62CEJ3MZHNHR6","created_at":"2026-07-05T11:34:11.260046+00:00"},{"alias_kind":"pith_short_8","alias_value":"L4U62CEJ","created_at":"2026-07-05T11:34:11.260046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20682","citing_title":"IndusAgent: Reinforcing Open-Vocabulary Industrial Anomaly Detection with Agentic Tools","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5","json":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5.json","graph_json":"https://pith.science/api/pith-number/L4U62CEJ3MZHNHR6QVT3XHN3Q5/graph.json","events_json":"https://pith.science/api/pith-number/L4U62CEJ3MZHNHR6QVT3XHN3Q5/events.json","paper":"https://pith.science/paper/L4U62CEJ"},"agent_actions":{"view_html":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5","download_json":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5.json","view_paper":"https://pith.science/paper/L4U62CEJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.03253&json=true","fetch_graph":"https://pith.science/api/pith-number/L4U62CEJ3MZHNHR6QVT3XHN3Q5/graph.json","fetch_events":"https://pith.science/api/pith-number/L4U62CEJ3MZHNHR6QVT3XHN3Q5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5/action/storage_attestation","attest_author":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5/action/author_attestation","sign_citation":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5/action/citation_signature","submit_replication":"https://pith.science/pith/L4U62CEJ3MZHNHR6QVT3XHN3Q5/action/replication_record"}},"created_at":"2026-07-05T11:34:11.260046+00:00","updated_at":"2026-07-05T11:34:11.260046+00:00"}