{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:HBW654ZOP7C2ELIXNZJOUP2AHG","short_pith_number":"pith:HBW654ZO","schema_version":"1.0","canonical_sha256":"386deef32e7fc5a22d176e52ea3f4039af921d5e9cb51a16d75986f94b93f827","source":{"kind":"arxiv","id":"2008.09470","version":1},"attestation_state":"computed","paper":{"title":"Top2Vec: Distributed Representations of Topics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Dimo Angelov","submitted_at":"2020-08-19T20:58:27Z","abstract_excerpt":"Topic modeling is used for discovering latent semantic structure, usually referred to as topics, in a large collection of documents. The most widely used methods are Latent Dirichlet Allocation and Probabilistic Latent Semantic Analysis. Despite their popularity they have several weaknesses. In order to achieve optimal results they often require the number of topics to be known, custom stop-word lists, stemming, and lemmatization. Additionally these methods rely on bag-of-words representation of documents which ignore the ordering and semantics of words. Distributed representations of document"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2008.09470","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-08-19T20:58:27Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"35870eb72130e260a84c4a54222736695b69f43ac3158351ff73b7f484e79181","abstract_canon_sha256":"5b347e496eb41e881f9c364940dc71b994010e4227c40ec6aad64c2ea7e0d839"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:28:55.228634Z","signature_b64":"lM9sK0Y128TN7EwK17DN17XIYnDMmBHd9FV2L4B4J/Lgvdt4Sh8wEX4dqukPl4vw1u9VjMOBoa1zWuwDXZxKDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"386deef32e7fc5a22d176e52ea3f4039af921d5e9cb51a16d75986f94b93f827","last_reissued_at":"2026-07-05T01:28:55.228204Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:28:55.228204Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Top2Vec: Distributed Representations of Topics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Dimo Angelov","submitted_at":"2020-08-19T20:58:27Z","abstract_excerpt":"Topic modeling is used for discovering latent semantic structure, usually referred to as topics, in a large collection of documents. The most widely used methods are Latent Dirichlet Allocation and Probabilistic Latent Semantic Analysis. Despite their popularity they have several weaknesses. In order to achieve optimal results they often require the number of topics to be known, custom stop-word lists, stemming, and lemmatization. Additionally these methods rely on bag-of-words representation of documents which ignore the ordering and semantics of words. Distributed representations of document"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.09470","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2008.09470/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2008.09470","created_at":"2026-07-05T01:28:55.228260+00:00"},{"alias_kind":"arxiv_version","alias_value":"2008.09470v1","created_at":"2026-07-05T01:28:55.228260+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.09470","created_at":"2026-07-05T01:28:55.228260+00:00"},{"alias_kind":"pith_short_12","alias_value":"HBW654ZOP7C2","created_at":"2026-07-05T01:28:55.228260+00:00"},{"alias_kind":"pith_short_16","alias_value":"HBW654ZOP7C2ELIX","created_at":"2026-07-05T01:28:55.228260+00:00"},{"alias_kind":"pith_short_8","alias_value":"HBW654ZO","created_at":"2026-07-05T01:28:55.228260+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26845","citing_title":"A Shared IPTC Topic Space for Cross-Source Topic Modelling","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01833","citing_title":"Non-synchronism in Global Usage of Research Methods in Library and Information Science from 1990 to 2019","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01828","citing_title":"Gender Differences in Research Topic and Method Selection in Library and Information Science: Perspectives from Three Top Journals","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11336","citing_title":"Much of Geospatial Web Search Is Beyond Traditional GIS","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20597","citing_title":"From Punishment to Protection: Charting Six Decades of U.S. Juvenile Justice Through Topic Modeling and LLM-Assisted Analysis","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25320","citing_title":"Data-Driven Evolution of Library and Information Science Research Methods (1990-2022): A Perspective Based on Fine-grained Method Entities","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23093","citing_title":"A Comparative Evaluation of Structural Topic Models and BERTopic for Short, Open-Ended Survey Responses","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18752","citing_title":"Traditional statistical representations outperform generative AI in identifying expert peer reviewers","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2512.17795","citing_title":"Intelligent Knowledge Mining Framework: Bridging AI Analysis and Trustworthy Preservation","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13521","citing_title":"Granite Embedding Multilingual R2 Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03180","citing_title":"PRISM: LLM-Guided Semantic Clustering for High-Precision Topics","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11336","citing_title":"Much of Geospatial Web Search Is Beyond Traditional GIS","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20528","citing_title":"Evolution of Research Method Usage Across the Academic Careers of Library and Information Science Scholars","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07562","citing_title":"Reasoning-Based Refinement of Unsupervised Text Clusters with LLMs","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG","json":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG.json","graph_json":"https://pith.science/api/pith-number/HBW654ZOP7C2ELIXNZJOUP2AHG/graph.json","events_json":"https://pith.science/api/pith-number/HBW654ZOP7C2ELIXNZJOUP2AHG/events.json","paper":"https://pith.science/paper/HBW654ZO"},"agent_actions":{"view_html":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG","download_json":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG.json","view_paper":"https://pith.science/paper/HBW654ZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2008.09470&json=true","fetch_graph":"https://pith.science/api/pith-number/HBW654ZOP7C2ELIXNZJOUP2AHG/graph.json","fetch_events":"https://pith.science/api/pith-number/HBW654ZOP7C2ELIXNZJOUP2AHG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG/action/storage_attestation","attest_author":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG/action/author_attestation","sign_citation":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG/action/citation_signature","submit_replication":"https://pith.science/pith/HBW654ZOP7C2ELIXNZJOUP2AHG/action/replication_record"}},"created_at":"2026-07-05T01:28:55.228260+00:00","updated_at":"2026-07-05T01:28:55.228260+00:00"}