{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MZMJNZX4CR6A27KOBOXSHJC2AG","short_pith_number":"pith:MZMJNZX4","schema_version":"1.0","canonical_sha256":"665896e6fc147c0d7d4e0baf23a45a0184280dec3b607bbd59adca0d98fd49cc","source":{"kind":"arxiv","id":"2401.02709","version":1},"attestation_state":"computed","paper":{"title":"German Text Embedding Clustering Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bert Arnrich, Christopher Irrgang, Silvan Wehrli","submitted_at":"2024-01-05T08:42:45Z","abstract_excerpt":"This work introduces a benchmark assessing the performance of clustering German text embeddings in different domains. This benchmark is driven by the increasing use of clustering neural text embeddings in tasks that require the grouping of texts (such as topic modeling) and the need for German resources in existing benchmarks. We provide an initial analysis for a range of pre-trained mono- and multilingual models evaluated on the outcome of different clustering algorithms. Results include strong performing mono- and multilingual models. Reducing the dimensions of embeddings can further improve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.02709","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-05T08:42:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0c06bb21ffaa77601b665f7d37d2634e9ab13e4c4592e1e1201afdb15c47d78e","abstract_canon_sha256":"c50b220a8170683d6cfbd3601ef6f725134f12ea0dd2af7d3ab8754382c8fd65"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:30:30.173396Z","signature_b64":"TuqXkWLvIUpmFotoOOf5i3hFU26QNHecnJp9MPOM7VWxWtFk3qm/ILe6Xwughq+zeOvWx9DT/L7ClIRaskTMDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"665896e6fc147c0d7d4e0baf23a45a0184280dec3b607bbd59adca0d98fd49cc","last_reissued_at":"2026-07-05T07:30:30.173050Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:30:30.173050Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"German Text Embedding Clustering Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bert Arnrich, Christopher Irrgang, Silvan Wehrli","submitted_at":"2024-01-05T08:42:45Z","abstract_excerpt":"This work introduces a benchmark assessing the performance of clustering German text embeddings in different domains. This benchmark is driven by the increasing use of clustering neural text embeddings in tasks that require the grouping of texts (such as topic modeling) and the need for German resources in existing benchmarks. We provide an initial analysis for a range of pre-trained mono- and multilingual models evaluated on the outcome of different clustering algorithms. Results include strong performing mono- and multilingual models. Reducing the dimensions of embeddings can further improve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.02709","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.02709/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.02709","created_at":"2026-07-05T07:30:30.173109+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.02709v1","created_at":"2026-07-05T07:30:30.173109+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02709","created_at":"2026-07-05T07:30:30.173109+00:00"},{"alias_kind":"pith_short_12","alias_value":"MZMJNZX4CR6A","created_at":"2026-07-05T07:30:30.173109+00:00"},{"alias_kind":"pith_short_16","alias_value":"MZMJNZX4CR6A27KO","created_at":"2026-07-05T07:30:30.173109+00:00"},{"alias_kind":"pith_short_8","alias_value":"MZMJNZX4","created_at":"2026-07-05T07:30:30.173109+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21182","citing_title":"Maintaining MTEB: Towards Long Term Usability and Reproducibility of Embedding Benchmarks","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG","json":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG.json","graph_json":"https://pith.science/api/pith-number/MZMJNZX4CR6A27KOBOXSHJC2AG/graph.json","events_json":"https://pith.science/api/pith-number/MZMJNZX4CR6A27KOBOXSHJC2AG/events.json","paper":"https://pith.science/paper/MZMJNZX4"},"agent_actions":{"view_html":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG","download_json":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG.json","view_paper":"https://pith.science/paper/MZMJNZX4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.02709&json=true","fetch_graph":"https://pith.science/api/pith-number/MZMJNZX4CR6A27KOBOXSHJC2AG/graph.json","fetch_events":"https://pith.science/api/pith-number/MZMJNZX4CR6A27KOBOXSHJC2AG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG/action/storage_attestation","attest_author":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG/action/author_attestation","sign_citation":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG/action/citation_signature","submit_replication":"https://pith.science/pith/MZMJNZX4CR6A27KOBOXSHJC2AG/action/replication_record"}},"created_at":"2026-07-05T07:30:30.173109+00:00","updated_at":"2026-07-05T07:30:30.173109+00:00"}