{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M2SK5SHFHK6GO422PLTJZ3YOYR","short_pith_number":"pith:M2SK5SHF","schema_version":"1.0","canonical_sha256":"66a4aec8e53abc67735a7ae69cef0ec46a7dcfd29bda06448590714d315c3c4f","source":{"kind":"arxiv","id":"2409.15626","version":1},"attestation_state":"computed","paper":{"title":"Qualitative Insights Tool (QualIT): LLM Enhanced Topic Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Alex Gil, Anshul Mittal, Rutu Mulkar, Satya Kapoor, Sreyoshi Bhaduri","submitted_at":"2024-09-24T00:09:41Z","abstract_excerpt":"Topic modeling is a widely used technique for uncovering thematic structures from large text corpora. However, most topic modeling approaches e.g. Latent Dirichlet Allocation (LDA) struggle to capture nuanced semantics and contextual understanding required to accurately model complex narratives. Recent advancements in this area include methods like BERTopic, which have demonstrated significantly improved topic coherence and thus established a new standard for benchmarking. In this paper, we present a novel approach, the Qualitative Insights Tool (QualIT) that integrates large language models ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.15626","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-09-24T00:09:41Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"cf6fc7fdac535a57c22e3c2cc60005a6590d2a89534dd6bdc82261cc5e9f1f23","abstract_canon_sha256":"911612b97b19c03d27a5df12d711c619fe0f3cbb563c07868b040e7d82e75e36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:11:17.466344Z","signature_b64":"WPq+6/BrvRXDbyenAGnD9jXywTqFV+i0N8m+F//ZKZhcqqy+rthPYP5+SF9Em3EOj15DsZVIM+mCVZj2lxi4BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66a4aec8e53abc67735a7ae69cef0ec46a7dcfd29bda06448590714d315c3c4f","last_reissued_at":"2026-07-05T09:11:17.465835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:11:17.465835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Qualitative Insights Tool (QualIT): LLM Enhanced Topic Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Alex Gil, Anshul Mittal, Rutu Mulkar, Satya Kapoor, Sreyoshi Bhaduri","submitted_at":"2024-09-24T00:09:41Z","abstract_excerpt":"Topic modeling is a widely used technique for uncovering thematic structures from large text corpora. However, most topic modeling approaches e.g. Latent Dirichlet Allocation (LDA) struggle to capture nuanced semantics and contextual understanding required to accurately model complex narratives. Recent advancements in this area include methods like BERTopic, which have demonstrated significantly improved topic coherence and thus established a new standard for benchmarking. In this paper, we present a novel approach, the Qualitative Insights Tool (QualIT) that integrates large language models ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15626","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.15626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.15626","created_at":"2026-07-05T09:11:17.465897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.15626v1","created_at":"2026-07-05T09:11:17.465897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15626","created_at":"2026-07-05T09:11:17.465897+00:00"},{"alias_kind":"pith_short_12","alias_value":"M2SK5SHFHK6G","created_at":"2026-07-05T09:11:17.465897+00:00"},{"alias_kind":"pith_short_16","alias_value":"M2SK5SHFHK6GO422","created_at":"2026-07-05T09:11:17.465897+00:00"},{"alias_kind":"pith_short_8","alias_value":"M2SK5SHF","created_at":"2026-07-05T09:11:17.465897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19174","citing_title":"Restructure This: Using AI to Restructure Onboarding Documents to Reduce Cognitive Overload","ref_index":106,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR","json":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR.json","graph_json":"https://pith.science/api/pith-number/M2SK5SHFHK6GO422PLTJZ3YOYR/graph.json","events_json":"https://pith.science/api/pith-number/M2SK5SHFHK6GO422PLTJZ3YOYR/events.json","paper":"https://pith.science/paper/M2SK5SHF"},"agent_actions":{"view_html":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR","download_json":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR.json","view_paper":"https://pith.science/paper/M2SK5SHF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.15626&json=true","fetch_graph":"https://pith.science/api/pith-number/M2SK5SHFHK6GO422PLTJZ3YOYR/graph.json","fetch_events":"https://pith.science/api/pith-number/M2SK5SHFHK6GO422PLTJZ3YOYR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR/action/storage_attestation","attest_author":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR/action/author_attestation","sign_citation":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR/action/citation_signature","submit_replication":"https://pith.science/pith/M2SK5SHFHK6GO422PLTJZ3YOYR/action/replication_record"}},"created_at":"2026-07-05T09:11:17.465897+00:00","updated_at":"2026-07-05T09:11:17.465897+00:00"}