{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:FPAKMZFMCQBX7ZZM7RMICHM7RZ","short_pith_number":"pith:FPAKMZFM","schema_version":"1.0","canonical_sha256":"2bc0a664ac14037fe72cfc58811d9f8e5d2e0450ec7f86ef39e90d2a142bc5b1","source":{"kind":"arxiv","id":"2107.02173","version":3},"attestation_state":"computed","paper":{"title":"Is Automated Topic Model Evaluation Broken?: The Incoherence of Coherence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Hoyle, Andrew Hian-Cheong, Denis Peskov, Jordan Boyd-Graber, Philip Resnik, Pranav Goel","submitted_at":"2021-07-05T17:58:52Z","abstract_excerpt":"Topic model evaluation, like evaluation of other unsupervised methods, can be contentious. However, the field has coalesced around automated estimates of topic coherence, which rely on the frequency of word co-occurrences in a reference corpus. Contemporary neural topic models surpass classical ones according to these metrics. At the same time, topic model evaluation suffers from a validation gap: automated coherence, developed for classical models, has not been validated using human experimentation for neural models. In addition, a meta-analysis of topic modeling literature reveals a substant"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.02173","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-07-05T17:58:52Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"573f30a59bdef35b91501b7185cc7b08c94b324bbc129987a60f62286812143f","abstract_canon_sha256":"687b827f7018ca8f6e27877aee40ec07b88edc82d90610693e8f999b852f1aba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:26:42.565432Z","signature_b64":"JwYziLx1NVWe3ulfNz0EaQ/L6HOrHI3gHMPRc9M4wBSK1gn88aQ3TGgHDIBRoTmKXYP28NerZf+RTGL6RG3BCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2bc0a664ac14037fe72cfc58811d9f8e5d2e0450ec7f86ef39e90d2a142bc5b1","last_reissued_at":"2026-07-05T03:26:42.564932Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:26:42.564932Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Automated Topic Model Evaluation Broken?: The Incoherence of Coherence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Hoyle, Andrew Hian-Cheong, Denis Peskov, Jordan Boyd-Graber, Philip Resnik, Pranav Goel","submitted_at":"2021-07-05T17:58:52Z","abstract_excerpt":"Topic model evaluation, like evaluation of other unsupervised methods, can be contentious. However, the field has coalesced around automated estimates of topic coherence, which rely on the frequency of word co-occurrences in a reference corpus. Contemporary neural topic models surpass classical ones according to these metrics. At the same time, topic model evaluation suffers from a validation gap: automated coherence, developed for classical models, has not been validated using human experimentation for neural models. In addition, a meta-analysis of topic modeling literature reveals a substant"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.02173","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.02173/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.02173","created_at":"2026-07-05T03:26:42.564988+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.02173v3","created_at":"2026-07-05T03:26:42.564988+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.02173","created_at":"2026-07-05T03:26:42.564988+00:00"},{"alias_kind":"pith_short_12","alias_value":"FPAKMZFMCQBX","created_at":"2026-07-05T03:26:42.564988+00:00"},{"alias_kind":"pith_short_16","alias_value":"FPAKMZFMCQBX7ZZM","created_at":"2026-07-05T03:26:42.564988+00:00"},{"alias_kind":"pith_short_8","alias_value":"FPAKMZFM","created_at":"2026-07-05T03:26:42.564988+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.04761","citing_title":"Cognitive Twins: Investigating Personalized Thinking Model Building and Its Performance Enhancement with Human-in-the-Loop","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ","json":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ.json","graph_json":"https://pith.science/api/pith-number/FPAKMZFMCQBX7ZZM7RMICHM7RZ/graph.json","events_json":"https://pith.science/api/pith-number/FPAKMZFMCQBX7ZZM7RMICHM7RZ/events.json","paper":"https://pith.science/paper/FPAKMZFM"},"agent_actions":{"view_html":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ","download_json":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ.json","view_paper":"https://pith.science/paper/FPAKMZFM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.02173&json=true","fetch_graph":"https://pith.science/api/pith-number/FPAKMZFMCQBX7ZZM7RMICHM7RZ/graph.json","fetch_events":"https://pith.science/api/pith-number/FPAKMZFMCQBX7ZZM7RMICHM7RZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ/action/storage_attestation","attest_author":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ/action/author_attestation","sign_citation":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ/action/citation_signature","submit_replication":"https://pith.science/pith/FPAKMZFMCQBX7ZZM7RMICHM7RZ/action/replication_record"}},"created_at":"2026-07-05T03:26:42.564988+00:00","updated_at":"2026-07-05T03:26:42.564988+00:00"}