{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:JSIKLILLDTUGWVCDHATUJUM3PU","short_pith_number":"pith:JSIKLILL","schema_version":"1.0","canonical_sha256":"4c90a5a16b1ce86b5443382744d19b7d2c3f1ca8bc4a86fb6a2edbf3176263c1","source":{"kind":"arxiv","id":"2010.11821","version":3},"attestation_state":"computed","paper":{"title":"Scalable Hierarchical Agglomerative Clustering","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amr Ahmed, Andrew McCallum, Avinava Dubey, Bryon Tjanaka, Gokhan Mergen, Guru Guruganesh, Manzil Zaheer, Marc Najork, Mert Terzihan, Nicholas Monath, Yuan Wang, Yuchen Wu","submitted_at":"2020-10-22T15:58:35Z","abstract_excerpt":"The applicability of agglomerative clustering, for inferring both hierarchical and flat clustering, is limited by its scalability. Existing scalable hierarchical clustering methods sacrifice quality for speed and often lead to over-merging of clusters. In this paper, we present a scalable, agglomerative method for hierarchical clustering that does not sacrifice quality and scales to billions of data points. We perform a detailed theoretical analysis, showing that under mild separability conditions our algorithm can not only recover the optimal flat partition, but also provide a two-approximati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.11821","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2020-10-22T15:58:35Z","cross_cats_sorted":[],"title_canon_sha256":"80f3cdfdadeb3a34927a063fdde23b619ab17368e113ec3d36690fafbd933a8c","abstract_canon_sha256":"e6f05299b96f239ad423ad98bacc74e297fade3e0877d20824c5f9857436dced"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:18:53.893592Z","signature_b64":"e5TfAqFgB23i6CnmqjryaaAVhWp70+xeOnJ4yh+ws7mP7rTFKx/o51rwegcuW7kJnKvjVXfkrBoNcNEOO2GeCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c90a5a16b1ce86b5443382744d19b7d2c3f1ca8bc4a86fb6a2edbf3176263c1","last_reissued_at":"2026-07-05T03:18:53.893188Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:18:53.893188Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable Hierarchical Agglomerative Clustering","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amr Ahmed, Andrew McCallum, Avinava Dubey, Bryon Tjanaka, Gokhan Mergen, Guru Guruganesh, Manzil Zaheer, Marc Najork, Mert Terzihan, Nicholas Monath, Yuan Wang, Yuchen Wu","submitted_at":"2020-10-22T15:58:35Z","abstract_excerpt":"The applicability of agglomerative clustering, for inferring both hierarchical and flat clustering, is limited by its scalability. Existing scalable hierarchical clustering methods sacrifice quality for speed and often lead to over-merging of clusters. In this paper, we present a scalable, agglomerative method for hierarchical clustering that does not sacrifice quality and scales to billions of data points. We perform a detailed theoretical analysis, showing that under mild separability conditions our algorithm can not only recover the optimal flat partition, but also provide a two-approximati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.11821","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.11821/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.11821","created_at":"2026-07-05T03:18:53.893248+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.11821v3","created_at":"2026-07-05T03:18:53.893248+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.11821","created_at":"2026-07-05T03:18:53.893248+00:00"},{"alias_kind":"pith_short_12","alias_value":"JSIKLILLDTUG","created_at":"2026-07-05T03:18:53.893248+00:00"},{"alias_kind":"pith_short_16","alias_value":"JSIKLILLDTUGWVCD","created_at":"2026-07-05T03:18:53.893248+00:00"},{"alias_kind":"pith_short_8","alias_value":"JSIKLILL","created_at":"2026-07-05T03:18:53.893248+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU","json":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU.json","graph_json":"https://pith.science/api/pith-number/JSIKLILLDTUGWVCDHATUJUM3PU/graph.json","events_json":"https://pith.science/api/pith-number/JSIKLILLDTUGWVCDHATUJUM3PU/events.json","paper":"https://pith.science/paper/JSIKLILL"},"agent_actions":{"view_html":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU","download_json":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU.json","view_paper":"https://pith.science/paper/JSIKLILL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.11821&json=true","fetch_graph":"https://pith.science/api/pith-number/JSIKLILLDTUGWVCDHATUJUM3PU/graph.json","fetch_events":"https://pith.science/api/pith-number/JSIKLILLDTUGWVCDHATUJUM3PU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU/action/storage_attestation","attest_author":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU/action/author_attestation","sign_citation":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU/action/citation_signature","submit_replication":"https://pith.science/pith/JSIKLILLDTUGWVCDHATUJUM3PU/action/replication_record"}},"created_at":"2026-07-05T03:18:53.893248+00:00","updated_at":"2026-07-05T03:18:53.893248+00:00"}