{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NI7UYCBAT5E32KLYO2MBB4YCUI","short_pith_number":"pith:NI7UYCBA","schema_version":"1.0","canonical_sha256":"6a3f4c08209f49bd2978769810f302a22cd2266e21be27bc14559029394110f3","source":{"kind":"arxiv","id":"2408.08310","version":1},"attestation_state":"computed","paper":{"title":"ScalingFilter: Assessing Data Quality through Inverse Utilization of Scaling Laws","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Han Hu, Houwen Peng, Miaosen Zhang, Nenghai Yu, Ruihang Li, Yixuan Wei","submitted_at":"2024-08-15T17:59:30Z","abstract_excerpt":"High-quality data is crucial for the pre-training performance of large language models. Unfortunately, existing quality filtering methods rely on a known high-quality dataset as reference, which can introduce potential bias and compromise diversity. In this paper, we propose ScalingFilter, a novel approach that evaluates text quality based on the perplexity difference between two language models trained on the same data, thereby eliminating the influence of the reference dataset in the filtering process. An theoretical analysis shows that ScalingFilter is equivalent to an inverse utilization o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.08310","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-15T17:59:30Z","cross_cats_sorted":[],"title_canon_sha256":"c1aa6008c821c97b01c257d645885ffefff453c46f4a12c7c36c34c84a510af7","abstract_canon_sha256":"b28e9e8f8c23219a9a70075433f3f55d136917f0638b93423a7cfa9ad55b7e9e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:47.000146Z","signature_b64":"VGZP+GkEJc6l4StGRLFAYUmQfA7yjQ8eOvC7NoOAl+izndg/l+dktAGCOdR5h1ZE4wy2Ve48yFq+imUaetxLCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a3f4c08209f49bd2978769810f302a22cd2266e21be27bc14559029394110f3","last_reissued_at":"2026-07-05T08:55:46.999735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:46.999735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ScalingFilter: Assessing Data Quality through Inverse Utilization of Scaling Laws","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Han Hu, Houwen Peng, Miaosen Zhang, Nenghai Yu, Ruihang Li, Yixuan Wei","submitted_at":"2024-08-15T17:59:30Z","abstract_excerpt":"High-quality data is crucial for the pre-training performance of large language models. Unfortunately, existing quality filtering methods rely on a known high-quality dataset as reference, which can introduce potential bias and compromise diversity. In this paper, we propose ScalingFilter, a novel approach that evaluates text quality based on the perplexity difference between two language models trained on the same data, thereby eliminating the influence of the reference dataset in the filtering process. An theoretical analysis shows that ScalingFilter is equivalent to an inverse utilization o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.08310","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.08310/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.08310","created_at":"2026-07-05T08:55:46.999791+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.08310v1","created_at":"2026-07-05T08:55:46.999791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.08310","created_at":"2026-07-05T08:55:46.999791+00:00"},{"alias_kind":"pith_short_12","alias_value":"NI7UYCBAT5E3","created_at":"2026-07-05T08:55:46.999791+00:00"},{"alias_kind":"pith_short_16","alias_value":"NI7UYCBAT5E32KLY","created_at":"2026-07-05T08:55:46.999791+00:00"},{"alias_kind":"pith_short_8","alias_value":"NI7UYCBA","created_at":"2026-07-05T08:55:46.999791+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10142","citing_title":"DB-3DME: From Dataset to Benchmark for Human-aligned Automatic 3D Mesh Evaluation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07546","citing_title":"On the Invariance and Generality of Neural Scaling Laws","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI","json":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI.json","graph_json":"https://pith.science/api/pith-number/NI7UYCBAT5E32KLYO2MBB4YCUI/graph.json","events_json":"https://pith.science/api/pith-number/NI7UYCBAT5E32KLYO2MBB4YCUI/events.json","paper":"https://pith.science/paper/NI7UYCBA"},"agent_actions":{"view_html":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI","download_json":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI.json","view_paper":"https://pith.science/paper/NI7UYCBA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.08310&json=true","fetch_graph":"https://pith.science/api/pith-number/NI7UYCBAT5E32KLYO2MBB4YCUI/graph.json","fetch_events":"https://pith.science/api/pith-number/NI7UYCBAT5E32KLYO2MBB4YCUI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI/action/storage_attestation","attest_author":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI/action/author_attestation","sign_citation":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI/action/citation_signature","submit_replication":"https://pith.science/pith/NI7UYCBAT5E32KLYO2MBB4YCUI/action/replication_record"}},"created_at":"2026-07-05T08:55:46.999791+00:00","updated_at":"2026-07-05T08:55:46.999791+00:00"}