{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KNT7YGSANRTKYAQKH2EH46FQCL","short_pith_number":"pith:KNT7YGSA","schema_version":"1.0","canonical_sha256":"5367fc1a406c66ac020a3e887e78b012fb3b39ecef7f42184bd75b7894d0c3b7","source":{"kind":"arxiv","id":"2502.06007","version":1},"attestation_state":"computed","paper":{"title":"Transformers versus the EM Algorithm in Multi-class Clustering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Han Liu, Hong-Yu Chen, Jianqing Fan, Yihan He, Yuan Cao","submitted_at":"2025-02-09T19:51:58Z","abstract_excerpt":"LLMs demonstrate significant inference capacities in complicated machine learning tasks, using the Transformer model as its backbone. Motivated by the limited understanding of such models on the unsupervised learning problems, we study the learning guarantees of Transformers in performing multi-class clustering of the Gaussian Mixture Models. We develop a theory drawing strong connections between the Softmax Attention layers and the workflow of the EM algorithm on clustering the mixture of Gaussians. Our theory provides approximation bounds for the Expectation and Maximization steps by proving"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06007","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2025-02-09T19:51:58Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2b28bec24991a18e965a1a71008eb186ab71161262445a34a25709b4cf56725c","abstract_canon_sha256":"444962ffd60d6286d1b9c4847754a12d8c0175f54f33e1c4caa9e710e142192e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:46.343436Z","signature_b64":"uS4fDSDEEUd8Gkc5On//X1h2ZIeMWLxnDECLTTHLOEz5TQ3BXwSZPlyEvyFbeTBORThLaf0EsZeAEZ/0eu5OCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5367fc1a406c66ac020a3e887e78b012fb3b39ecef7f42184bd75b7894d0c3b7","last_reissued_at":"2026-07-05T10:11:46.342889Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:46.342889Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers versus the EM Algorithm in Multi-class Clustering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Han Liu, Hong-Yu Chen, Jianqing Fan, Yihan He, Yuan Cao","submitted_at":"2025-02-09T19:51:58Z","abstract_excerpt":"LLMs demonstrate significant inference capacities in complicated machine learning tasks, using the Transformer model as its backbone. Motivated by the limited understanding of such models on the unsupervised learning problems, we study the learning guarantees of Transformers in performing multi-class clustering of the Gaussian Mixture Models. We develop a theory drawing strong connections between the Softmax Attention layers and the workflow of the EM algorithm on clustering the mixture of Gaussians. Our theory provides approximation bounds for the Expectation and Maximization steps by proving"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06007","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06007/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06007","created_at":"2026-07-05T10:11:46.342971+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06007v1","created_at":"2026-07-05T10:11:46.342971+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06007","created_at":"2026-07-05T10:11:46.342971+00:00"},{"alias_kind":"pith_short_12","alias_value":"KNT7YGSANRTK","created_at":"2026-07-05T10:11:46.342971+00:00"},{"alias_kind":"pith_short_16","alias_value":"KNT7YGSANRTKYAQK","created_at":"2026-07-05T10:11:46.342971+00:00"},{"alias_kind":"pith_short_8","alias_value":"KNT7YGSA","created_at":"2026-07-05T10:11:46.342971+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06609","citing_title":"Transformers Efficiently Perform In-Context Logistic Regression via Normalized Gradient Descent","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL","json":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL.json","graph_json":"https://pith.science/api/pith-number/KNT7YGSANRTKYAQKH2EH46FQCL/graph.json","events_json":"https://pith.science/api/pith-number/KNT7YGSANRTKYAQKH2EH46FQCL/events.json","paper":"https://pith.science/paper/KNT7YGSA"},"agent_actions":{"view_html":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL","download_json":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL.json","view_paper":"https://pith.science/paper/KNT7YGSA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06007&json=true","fetch_graph":"https://pith.science/api/pith-number/KNT7YGSANRTKYAQKH2EH46FQCL/graph.json","fetch_events":"https://pith.science/api/pith-number/KNT7YGSANRTKYAQKH2EH46FQCL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL/action/storage_attestation","attest_author":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL/action/author_attestation","sign_citation":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL/action/citation_signature","submit_replication":"https://pith.science/pith/KNT7YGSANRTKYAQKH2EH46FQCL/action/replication_record"}},"created_at":"2026-07-05T10:11:46.342971+00:00","updated_at":"2026-07-05T10:11:46.342971+00:00"}