{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LRGG5J3XZWQ7BMZPAKO56FD24N","short_pith_number":"pith:LRGG5J3X","schema_version":"1.0","canonical_sha256":"5c4c6ea777cda1f0b32f029ddf147ae3472435591fc0d3c0655f2bbb91a06e36","source":{"kind":"arxiv","id":"2309.02033","version":3},"attestation_state":"computed","paper":{"title":"Data-Juicer: A One-Stop Data Processing System for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DB","cs.DC"],"primary_cat":"cs.LG","authors_text":"Bolin Ding, Ce Ge, Daoyuan Chen, Dawei Gao, Hesen Chen, Jingren Zhou, Jinyang Gao, Xuchen Pan, Yaliang Li, Yilun Huang, Yuexiang Xie, Zhaoyang Liu, Zhijian Ma","submitted_at":"2023-09-05T08:22:07Z","abstract_excerpt":"The immense evolution in Large Language Models (LLMs) has underscored the importance of massive, heterogeneous, and high-quality data. A data recipe is a mixture of data from different sources for training LLMs, which plays a vital role in LLMs' performance. Existing open-source tools for LLM data processing are mostly tailored for specific data recipes. To continuously uncover the potential of LLMs, incorporate data from new sources, and improve LLMs' performance, we build a new system named Data-Juicer, with which we can efficiently generate diverse data recipes, explore different possibilit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.02033","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-05T08:22:07Z","cross_cats_sorted":["cs.DB","cs.DC"],"title_canon_sha256":"325f602546f62f64f34f51f46c50021a1c3840ae8d7f0c455c45a60e22e8fb4f","abstract_canon_sha256":"2d3688bce690dfeef8da38c3c8429f04b08ad05d8019b0beb67825c078a99f74"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:26:17.000748Z","signature_b64":"d89dJORufsmVDd6QCco3Bw2j4rxEhg1JuL+SWtGiiwxfg7XBHi6Aj6Q2GAkJ2GAEJDns0+Nq5Y1g1lRShZ1FCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c4c6ea777cda1f0b32f029ddf147ae3472435591fc0d3c0655f2bbb91a06e36","last_reissued_at":"2026-07-05T07:26:17.000226Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:26:17.000226Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Data-Juicer: A One-Stop Data Processing System for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DB","cs.DC"],"primary_cat":"cs.LG","authors_text":"Bolin Ding, Ce Ge, Daoyuan Chen, Dawei Gao, Hesen Chen, Jingren Zhou, Jinyang Gao, Xuchen Pan, Yaliang Li, Yilun Huang, Yuexiang Xie, Zhaoyang Liu, Zhijian Ma","submitted_at":"2023-09-05T08:22:07Z","abstract_excerpt":"The immense evolution in Large Language Models (LLMs) has underscored the importance of massive, heterogeneous, and high-quality data. A data recipe is a mixture of data from different sources for training LLMs, which plays a vital role in LLMs' performance. Existing open-source tools for LLM data processing are mostly tailored for specific data recipes. To continuously uncover the potential of LLMs, incorporate data from new sources, and improve LLMs' performance, we build a new system named Data-Juicer, with which we can efficiently generate diverse data recipes, explore different possibilit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.02033","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.02033/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.02033","created_at":"2026-07-05T07:26:17.000280+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.02033v3","created_at":"2026-07-05T07:26:17.000280+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.02033","created_at":"2026-07-05T07:26:17.000280+00:00"},{"alias_kind":"pith_short_12","alias_value":"LRGG5J3XZWQ7","created_at":"2026-07-05T07:26:17.000280+00:00"},{"alias_kind":"pith_short_16","alias_value":"LRGG5J3XZWQ7BMZP","created_at":"2026-07-05T07:26:17.000280+00:00"},{"alias_kind":"pith_short_8","alias_value":"LRGG5J3X","created_at":"2026-07-05T07:26:17.000280+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":266,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N","json":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N.json","graph_json":"https://pith.science/api/pith-number/LRGG5J3XZWQ7BMZPAKO56FD24N/graph.json","events_json":"https://pith.science/api/pith-number/LRGG5J3XZWQ7BMZPAKO56FD24N/events.json","paper":"https://pith.science/paper/LRGG5J3X"},"agent_actions":{"view_html":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N","download_json":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N.json","view_paper":"https://pith.science/paper/LRGG5J3X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.02033&json=true","fetch_graph":"https://pith.science/api/pith-number/LRGG5J3XZWQ7BMZPAKO56FD24N/graph.json","fetch_events":"https://pith.science/api/pith-number/LRGG5J3XZWQ7BMZPAKO56FD24N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N/action/storage_attestation","attest_author":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N/action/author_attestation","sign_citation":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N/action/citation_signature","submit_replication":"https://pith.science/pith/LRGG5J3XZWQ7BMZPAKO56FD24N/action/replication_record"}},"created_at":"2026-07-05T07:26:17.000280+00:00","updated_at":"2026-07-05T07:26:17.000280+00:00"}