{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2013:A3EA45VY3TQCKNBXIKPLJX2ED3","short_pith_number":"pith:A3EA45VY","schema_version":"1.0","canonical_sha256":"06c80e76b8dce0253437429eb4df441ed9e6c85ab39e62159a1cb19ae4bccfd0","source":{"kind":"arxiv","id":"1309.6874","version":1},"attestation_state":"computed","paper":{"title":"Integrating Document Clustering and Topic Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.IR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Eric P. Xing, Pengtao Xie","submitted_at":"2013-09-26T12:54:02Z","abstract_excerpt":"Document clustering and topic modeling are two closely related tasks which can mutually benefit each other. Topic modeling can project documents into a topic space which facilitates effective document clustering. Cluster labels discovered by document clustering can be incorporated into topic models to extract local topics specific to each cluster and global topics shared by all clusters. In this paper, we propose a multi-grain clustering topic model (MGCTM) which integrates document clustering and topic modeling into a unified framework and jointly performs the two tasks to achieve the overall"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1309.6874","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2013-09-26T12:54:02Z","cross_cats_sorted":["cs.CL","cs.IR","stat.ML"],"title_canon_sha256":"b1657a6299b3df26e8b45dca7c75747926929a67617f475ce957e1fdc316f5f5","abstract_canon_sha256":"790665de5c370c3460df0ba3d4065b610c094148d7e9c489d44e02f6f42c5fe2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T03:12:09.905611Z","signature_b64":"hZQb7piD/OVFH9gTQ9bPYQu2UH4d6/OkhwtdO1fOUNbZo5cCmloPxLJth2zhncfGJ5r6TNEBrN4X2XsisifTDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06c80e76b8dce0253437429eb4df441ed9e6c85ab39e62159a1cb19ae4bccfd0","last_reissued_at":"2026-05-18T03:12:09.904808Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T03:12:09.904808Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Integrating Document Clustering and Topic Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.IR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Eric P. Xing, Pengtao Xie","submitted_at":"2013-09-26T12:54:02Z","abstract_excerpt":"Document clustering and topic modeling are two closely related tasks which can mutually benefit each other. Topic modeling can project documents into a topic space which facilitates effective document clustering. Cluster labels discovered by document clustering can be incorporated into topic models to extract local topics specific to each cluster and global topics shared by all clusters. In this paper, we propose a multi-grain clustering topic model (MGCTM) which integrates document clustering and topic modeling into a unified framework and jointly performs the two tasks to achieve the overall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1309.6874","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1309.6874","created_at":"2026-05-18T03:12:09.904936+00:00"},{"alias_kind":"arxiv_version","alias_value":"1309.6874v1","created_at":"2026-05-18T03:12:09.904936+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1309.6874","created_at":"2026-05-18T03:12:09.904936+00:00"},{"alias_kind":"pith_short_12","alias_value":"A3EA45VY3TQC","created_at":"2026-05-18T12:27:38.830355+00:00"},{"alias_kind":"pith_short_16","alias_value":"A3EA45VY3TQCKNBX","created_at":"2026-05-18T12:27:38.830355+00:00"},{"alias_kind":"pith_short_8","alias_value":"A3EA45VY","created_at":"2026-05-18T12:27:38.830355+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.16046","citing_title":"Estimating the Effective Topics of Articles and journals Abstract Using LDA And K-Means Clustering Algorithm","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3","json":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3.json","graph_json":"https://pith.science/api/pith-number/A3EA45VY3TQCKNBXIKPLJX2ED3/graph.json","events_json":"https://pith.science/api/pith-number/A3EA45VY3TQCKNBXIKPLJX2ED3/events.json","paper":"https://pith.science/paper/A3EA45VY"},"agent_actions":{"view_html":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3","download_json":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3.json","view_paper":"https://pith.science/paper/A3EA45VY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1309.6874&json=true","fetch_graph":"https://pith.science/api/pith-number/A3EA45VY3TQCKNBXIKPLJX2ED3/graph.json","fetch_events":"https://pith.science/api/pith-number/A3EA45VY3TQCKNBXIKPLJX2ED3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3/action/storage_attestation","attest_author":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3/action/author_attestation","sign_citation":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3/action/citation_signature","submit_replication":"https://pith.science/pith/A3EA45VY3TQCKNBXIKPLJX2ED3/action/replication_record"}},"created_at":"2026-05-18T03:12:09.904936+00:00","updated_at":"2026-05-18T03:12:09.904936+00:00"}