{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:CNXCDZVSURIUAN7HLNNZKD5IGU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"da8a6b2dd2473986a3c6583e2735d46eacaea18c439ff6e1571d9746b1533ff6","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-06-19T17:36:34Z","title_canon_sha256":"decaa26bbf5a75cbd5f69d97a02beaaf32ae13bdd33c778baf1431b97b9c78c9"},"schema_version":"1.0","source":{"id":"2606.21631","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2606.21631","created_at":"2026-06-23T01:13:17Z"},{"alias_kind":"arxiv_version","alias_value":"2606.21631v1","created_at":"2026-06-23T01:13:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.21631","created_at":"2026-06-23T01:13:17Z"},{"alias_kind":"pith_short_12","alias_value":"CNXCDZVSURIU","created_at":"2026-06-23T01:13:17Z"},{"alias_kind":"pith_short_16","alias_value":"CNXCDZVSURIUAN7H","created_at":"2026-06-23T01:13:17Z"},{"alias_kind":"pith_short_8","alias_value":"CNXCDZVS","created_at":"2026-06-23T01:13:17Z"}],"graph_snapshots":[{"event_id":"sha256:35ccb2438b4117d2a8252efa88b8af042d53d201c3b72feae56d12b6f4a08bce","target":"graph","created_at":"2026-06-23T01:13:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2606.21631/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Data curation is a critical part of post-training pipelines for large language models, yet existing tools often treat ingestion, deduplication, synthetic generation, and quality filtering as separate stages. This fragmentation makes it difficult to audit pipeline decisions or understand why individual samples are rejected. CuratorKIT is an open-source Python library that covers this full lifecycle in a single configurable pipeline. The framework is composed of six source format readers and automatic schema detection, a pre-generation data hygiene layer for credentials, PII, and toxic content, ","authors_text":"Karun Sharma, Pratinav Seth, Soham Bhattacharjee, Vinay Kumar Sankarapu","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-06-19T17:36:34Z","title":"CuratorKIT : Data Curation and Synthetic Data Generation for LLM Post-Training"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.21631","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:757792a04278af227ef14c70cb4332489d00664dbf4b76751adee1e4d28fc780","target":"record","created_at":"2026-06-23T01:13:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"da8a6b2dd2473986a3c6583e2735d46eacaea18c439ff6e1571d9746b1533ff6","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-06-19T17:36:34Z","title_canon_sha256":"decaa26bbf5a75cbd5f69d97a02beaaf32ae13bdd33c778baf1431b97b9c78c9"},"schema_version":"1.0","source":{"id":"2606.21631","kind":"arxiv","version":1}},"canonical_sha256":"136e21e6b2a4514037e75b5b950fa8350241e8709fdfbfc7fbeb242966a7362b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"136e21e6b2a4514037e75b5b950fa8350241e8709fdfbfc7fbeb242966a7362b","first_computed_at":"2026-06-23T01:13:17.312137Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-23T01:13:17.312137Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"/utBePHwNxlkmc7nHh8ghz/79O7GVVud6hcJqryXU5wfFGtync9CxYz7NQJ+prNhw19nPuSzSO7b03F7DxDkBQ==","signature_status":"signed_v1","signed_at":"2026-06-23T01:13:17.312645Z","signed_message":"canonical_sha256_bytes"},"source_id":"2606.21631","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:757792a04278af227ef14c70cb4332489d00664dbf4b76751adee1e4d28fc780","sha256:35ccb2438b4117d2a8252efa88b8af042d53d201c3b72feae56d12b6f4a08bce"],"state_sha256":"c7ec081db83f6a3fdd82b0c2ee7ba8b75a8ed29df38f8c2b3b5678309f0d9b43"}