{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:YXIOPY5FV76OPU4JAZCVR2EA7B","short_pith_number":"pith:YXIOPY5F","canonical_record":{"source":{"id":"2606.28328","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.IR","submitted_at":"2026-05-17T16:52:44Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"70b374debe52eb3510857021c5713407631277c94a4b718799968d9ce4536fb9","abstract_canon_sha256":"699f45847b003b7e9059543e7697f95904ff57871ac1e9afd72f7ea99db0ea60"},"schema_version":"1.0"},"canonical_sha256":"c5d0e7e3a5affce7d389064558e880f841e087c2e0f057932c08d390f3a8d14b","source":{"kind":"arxiv","id":"2606.28328","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2606.28328","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"arxiv_version","alias_value":"2606.28328v1","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.28328","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"pith_short_12","alias_value":"YXIOPY5FV76O","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"pith_short_16","alias_value":"YXIOPY5FV76OPU4J","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"pith_short_8","alias_value":"YXIOPY5F","created_at":"2026-06-30T00:15:10Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:YXIOPY5FV76OPU4JAZCVR2EA7B","target":"record","payload":{"canonical_record":{"source":{"id":"2606.28328","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.IR","submitted_at":"2026-05-17T16:52:44Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"70b374debe52eb3510857021c5713407631277c94a4b718799968d9ce4536fb9","abstract_canon_sha256":"699f45847b003b7e9059543e7697f95904ff57871ac1e9afd72f7ea99db0ea60"},"schema_version":"1.0"},"canonical_sha256":"c5d0e7e3a5affce7d389064558e880f841e087c2e0f057932c08d390f3a8d14b","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-30T00:15:10.888811Z","signature_b64":"BdgAdwlN9OL3PP+uG3rJ33YfSp3HRHm7xpbOaPD0liI0cnnolAm2LyQ6iil2XaIoXVSSd7IeweJYVknXJRIpBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5d0e7e3a5affce7d389064558e880f841e087c2e0f057932c08d390f3a8d14b","last_reissued_at":"2026-06-30T00:15:10.888326Z","signature_status":"signed_v1","first_computed_at":"2026-06-30T00:15:10.888326Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2606.28328","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-30T00:15:10Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"efPGR2ifCJ5D9EMv6qsDvv/rC7LMbkRKhvMljNVtis+8uM/xrlcqGI1/9FVTGY1ezKJbeNoFf4ctghi/9mokDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-02T15:12:51.891147Z"},"content_sha256":"25d6e4a17def8c3787949ec9f605cfa7ac0e8c660a7e1e134cc6ef673c2a6942","schema_version":"1.0","event_id":"sha256:25d6e4a17def8c3787949ec9f605cfa7ac0e8c660a7e1e134cc6ef673c2a6942"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:YXIOPY5FV76OPU4JAZCVR2EA7B","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"TextClusterLab: An Integrated Framework for Reliable Text Clustering Studies","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.IR","authors_text":"Daoming Wan, Jimmy X. Huang, Yizheng Huang","submitted_at":"2026-05-17T16:52:44Z","abstract_excerpt":"In recent years, text clustering has become a critical technique for applications including intent discovery, topic mining, and recommendation systems. However, evaluating text clustering algorithms remains challenging since many real-world textual datasets are not suitable for clustering assessment due to ambiguous semantic boundaries, the high dimensionality of embeddings, and inconsistent cluster structure. Current clustering dataset generators are designed for numerical data, providing limited support for text-specific benchmarking. This paper introduces TextClusterLab, a comprehensive fra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.28328","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.28328/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-30T00:15:10Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"RvPAHKV/7Buz7jVYEm8kTH6lW9Sv5DgmRy6bOFTuvXcj0fpC2/oPYZeWV/YX4JjD69Ze9pBR3pkxIQvorqH6AA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-02T15:12:51.891532Z"},"content_sha256":"d4a5894f98b25d5c91fd8c25ab819531910d53ca7528ab2e67eae90a260b5b9a","schema_version":"1.0","event_id":"sha256:d4a5894f98b25d5c91fd8c25ab819531910d53ca7528ab2e67eae90a260b5b9a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YXIOPY5FV76OPU4JAZCVR2EA7B/bundle.json","state_url":"https://pith.science/pith/YXIOPY5FV76OPU4JAZCVR2EA7B/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YXIOPY5FV76OPU4JAZCVR2EA7B/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-02T15:12:51Z","links":{"resolver":"https://pith.science/pith/YXIOPY5FV76OPU4JAZCVR2EA7B","bundle":"https://pith.science/pith/YXIOPY5FV76OPU4JAZCVR2EA7B/bundle.json","state":"https://pith.science/pith/YXIOPY5FV76OPU4JAZCVR2EA7B/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YXIOPY5FV76OPU4JAZCVR2EA7B/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:YXIOPY5FV76OPU4JAZCVR2EA7B","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"699f45847b003b7e9059543e7697f95904ff57871ac1e9afd72f7ea99db0ea60","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.IR","submitted_at":"2026-05-17T16:52:44Z","title_canon_sha256":"70b374debe52eb3510857021c5713407631277c94a4b718799968d9ce4536fb9"},"schema_version":"1.0","source":{"id":"2606.28328","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2606.28328","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"arxiv_version","alias_value":"2606.28328v1","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.28328","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"pith_short_12","alias_value":"YXIOPY5FV76O","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"pith_short_16","alias_value":"YXIOPY5FV76OPU4J","created_at":"2026-06-30T00:15:10Z"},{"alias_kind":"pith_short_8","alias_value":"YXIOPY5F","created_at":"2026-06-30T00:15:10Z"}],"graph_snapshots":[{"event_id":"sha256:d4a5894f98b25d5c91fd8c25ab819531910d53ca7528ab2e67eae90a260b5b9a","target":"graph","created_at":"2026-06-30T00:15:10Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2606.28328/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In recent years, text clustering has become a critical technique for applications including intent discovery, topic mining, and recommendation systems. However, evaluating text clustering algorithms remains challenging since many real-world textual datasets are not suitable for clustering assessment due to ambiguous semantic boundaries, the high dimensionality of embeddings, and inconsistent cluster structure. Current clustering dataset generators are designed for numerical data, providing limited support for text-specific benchmarking. This paper introduces TextClusterLab, a comprehensive fra","authors_text":"Daoming Wan, Jimmy X. Huang, Yizheng Huang","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.IR","submitted_at":"2026-05-17T16:52:44Z","title":"TextClusterLab: An Integrated Framework for Reliable Text Clustering Studies"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.28328","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:25d6e4a17def8c3787949ec9f605cfa7ac0e8c660a7e1e134cc6ef673c2a6942","target":"record","created_at":"2026-06-30T00:15:10Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"699f45847b003b7e9059543e7697f95904ff57871ac1e9afd72f7ea99db0ea60","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.IR","submitted_at":"2026-05-17T16:52:44Z","title_canon_sha256":"70b374debe52eb3510857021c5713407631277c94a4b718799968d9ce4536fb9"},"schema_version":"1.0","source":{"id":"2606.28328","kind":"arxiv","version":1}},"canonical_sha256":"c5d0e7e3a5affce7d389064558e880f841e087c2e0f057932c08d390f3a8d14b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c5d0e7e3a5affce7d389064558e880f841e087c2e0f057932c08d390f3a8d14b","first_computed_at":"2026-06-30T00:15:10.888326Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-30T00:15:10.888326Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"BdgAdwlN9OL3PP+uG3rJ33YfSp3HRHm7xpbOaPD0liI0cnnolAm2LyQ6iil2XaIoXVSSd7IeweJYVknXJRIpBg==","signature_status":"signed_v1","signed_at":"2026-06-30T00:15:10.888811Z","signed_message":"canonical_sha256_bytes"},"source_id":"2606.28328","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:25d6e4a17def8c3787949ec9f605cfa7ac0e8c660a7e1e134cc6ef673c2a6942","sha256:d4a5894f98b25d5c91fd8c25ab819531910d53ca7528ab2e67eae90a260b5b9a"],"state_sha256":"7234384ddf03533fe9c6dbb68f7c157f8a341f9523e7d418ad1e3ba88d04362e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UV4QzbECPruUq15YjxwJ3Kv+0RopDuKnHHWldq7QalrAW0xLr3NgRYQmzdQAeo20H5+SJwCvPgwKunvPqTcSCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-02T15:12:51.893549Z","bundle_sha256":"fd749816aa95be444b3bca2a64d2ae6b14aaf1a2c4d73b36b451637d8ea850b3"}}