{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GN44QQHXIB4IMGMTJ3RKHZ52YE","short_pith_number":"pith:GN44QQHX","schema_version":"1.0","canonical_sha256":"3379c840f740788619934ee2a3e7bac10963c7361c9672507a0ad4acf07a818e","source":{"kind":"arxiv","id":"2106.15971","version":1},"attestation_state":"computed","paper":{"title":"Evaluation of Thematic Coherence in Microblogs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adam Tsakalidis, Bo Wang, Iman Munire Bilal, Maria Liakata, Rob Procter","submitted_at":"2021-06-30T10:32:59Z","abstract_excerpt":"Collecting together microblogs representing opinions about the same topics within the same timeframe is useful to a number of different tasks and practitioners. A major question is how to evaluate the quality of such thematic clusters. Here we create a corpus of microblog clusters from three different domains and time windows and define the task of evaluating thematic coherence. We provide annotation guidelines and human annotations of thematic coherence by journalist experts. We subsequently investigate the efficacy of different automated evaluation metrics for the task. We consider a range o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.15971","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-06-30T10:32:59Z","cross_cats_sorted":[],"title_canon_sha256":"65b520d3e05a630b72294adb652ffaed311723e14aed03ba6cd0ae9d5635a716","abstract_canon_sha256":"d58c30f32c2bd90243a4915c1d00c3ecc308bfff5ec9066e4267b2e833265bf7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:54:01.998175Z","signature_b64":"q/tagWirPuLUefyg4Ytncb4QePotqyp/sGRJ9UGfZya8mmJoFNLJk5d0uqcvUZj7W4lgd9oJCnGw0jikTmFABQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3379c840f740788619934ee2a3e7bac10963c7361c9672507a0ad4acf07a818e","last_reissued_at":"2026-07-05T02:54:01.997692Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:54:01.997692Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of Thematic Coherence in Microblogs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adam Tsakalidis, Bo Wang, Iman Munire Bilal, Maria Liakata, Rob Procter","submitted_at":"2021-06-30T10:32:59Z","abstract_excerpt":"Collecting together microblogs representing opinions about the same topics within the same timeframe is useful to a number of different tasks and practitioners. A major question is how to evaluate the quality of such thematic clusters. Here we create a corpus of microblog clusters from three different domains and time windows and define the task of evaluating thematic coherence. We provide annotation guidelines and human annotations of thematic coherence by journalist experts. We subsequently investigate the efficacy of different automated evaluation metrics for the task. We consider a range o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.15971","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.15971/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.15971","created_at":"2026-07-05T02:54:01.997743+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.15971v1","created_at":"2026-07-05T02:54:01.997743+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.15971","created_at":"2026-07-05T02:54:01.997743+00:00"},{"alias_kind":"pith_short_12","alias_value":"GN44QQHXIB4I","created_at":"2026-07-05T02:54:01.997743+00:00"},{"alias_kind":"pith_short_16","alias_value":"GN44QQHXIB4IMGMT","created_at":"2026-07-05T02:54:01.997743+00:00"},{"alias_kind":"pith_short_8","alias_value":"GN44QQHX","created_at":"2026-07-05T02:54:01.997743+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.08439","citing_title":"A document processing pipeline for the construction of a dataset for topic modeling based on the judgments of the Italian Supreme Court","ref_index":45,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE","json":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE.json","graph_json":"https://pith.science/api/pith-number/GN44QQHXIB4IMGMTJ3RKHZ52YE/graph.json","events_json":"https://pith.science/api/pith-number/GN44QQHXIB4IMGMTJ3RKHZ52YE/events.json","paper":"https://pith.science/paper/GN44QQHX"},"agent_actions":{"view_html":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE","download_json":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE.json","view_paper":"https://pith.science/paper/GN44QQHX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.15971&json=true","fetch_graph":"https://pith.science/api/pith-number/GN44QQHXIB4IMGMTJ3RKHZ52YE/graph.json","fetch_events":"https://pith.science/api/pith-number/GN44QQHXIB4IMGMTJ3RKHZ52YE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE/action/storage_attestation","attest_author":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE/action/author_attestation","sign_citation":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE/action/citation_signature","submit_replication":"https://pith.science/pith/GN44QQHXIB4IMGMTJ3RKHZ52YE/action/replication_record"}},"created_at":"2026-07-05T02:54:01.997743+00:00","updated_at":"2026-07-05T02:54:01.997743+00:00"}