{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2019:GAI2Y34JJOWP5UZIRSHLOYVQP4","short_pith_number":"pith:GAI2Y34J","canonical_record":{"source":{"id":"1901.09848","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-01-28T17:41:19Z","cross_cats_sorted":["cs.LG","physics.soc-ph"],"title_canon_sha256":"c608dd9a503bb1a15a51dd7a425568984ee8811a1b95ae0240105baf9263e8af","abstract_canon_sha256":"80eed141fbe4864c76331ed1f989053b85c064f97cf85d7b3c8974643241fdeb"},"schema_version":"1.0"},"canonical_sha256":"3011ac6f894bacfed3288c8eb762b07f265aa5e0aa7c540958f2bb29a1be15d2","source":{"kind":"arxiv","id":"1901.09848","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1901.09848","created_at":"2026-05-17T23:55:23Z"},{"alias_kind":"arxiv_version","alias_value":"1901.09848v1","created_at":"2026-05-17T23:55:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1901.09848","created_at":"2026-05-17T23:55:23Z"},{"alias_kind":"pith_short_12","alias_value":"GAI2Y34JJOWP","created_at":"2026-05-18T12:33:18Z"},{"alias_kind":"pith_short_16","alias_value":"GAI2Y34JJOWP5UZI","created_at":"2026-05-18T12:33:18Z"},{"alias_kind":"pith_short_8","alias_value":"GAI2Y34J","created_at":"2026-05-18T12:33:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2019:GAI2Y34JJOWP5UZIRSHLOYVQP4","target":"record","payload":{"canonical_record":{"source":{"id":"1901.09848","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-01-28T17:41:19Z","cross_cats_sorted":["cs.LG","physics.soc-ph"],"title_canon_sha256":"c608dd9a503bb1a15a51dd7a425568984ee8811a1b95ae0240105baf9263e8af","abstract_canon_sha256":"80eed141fbe4864c76331ed1f989053b85c064f97cf85d7b3c8974643241fdeb"},"schema_version":"1.0"},"canonical_sha256":"3011ac6f894bacfed3288c8eb762b07f265aa5e0aa7c540958f2bb29a1be15d2","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:55:23.674726Z","signature_b64":"qocYFJhdN+KoEOWOlEkd9ZFiRXm2fD+w/6a2VzOmrxvEocxs09mu3g9xaQB+jrZdKsXcfqxrVSVale/0GSqyDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3011ac6f894bacfed3288c8eb762b07f265aa5e0aa7c540958f2bb29a1be15d2","last_reissued_at":"2026-05-17T23:55:23.674253Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:55:23.674253Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1901.09848","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:55:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aSvC/BUekC9FKuHB1PhEJo9GwfK+R3DEGpOCKVlADBRd1MMOIOEzKVG6xv/sXFzOfgefkvsgZAE2afbLtDmZBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T04:23:05.755916Z"},"content_sha256":"3349c54694cdb2562c78b93e215650ec48320b7ef2837944036f64fba4e880cc","schema_version":"1.0","event_id":"sha256:3349c54694cdb2562c78b93e215650ec48320b7ef2837944036f64fba4e880cc"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2019:GAI2Y34JJOWP5UZIRSHLOYVQP4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"A new evaluation framework for topic modeling algorithms based on synthetic corpora","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","physics.soc-ph"],"primary_cat":"cs.CL","authors_text":"Doug Downey, Hanyu Shi, Isabel Diersen, Luis A. N. Amaral, Martin Gerlach","submitted_at":"2019-01-28T17:41:19Z","abstract_excerpt":"Topic models are in widespread use in natural language processing and beyond. Here, we propose a new framework for the evaluation of probabilistic topic modeling algorithms based on synthetic corpora containing an unambiguously defined ground truth topic structure. The major innovation of our approach is the ability to quantify the agreement between the planted and inferred topic structures by comparing the assigned topic labels at the level of the tokens. In experiments, our approach yields novel insights about the relative strengths of topic models as corpus characteristics vary, and the fir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1901.09848","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:55:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"OjVNzfH6VHIz2YXfDbJlzMPXmzvChNf3gENyY+me13GXxagnj0pMabLL1ajdMjYeADakjwC1lCDhnYcS2rttDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T04:23:05.756386Z"},"content_sha256":"39a5dac9a7cfe8320a7886977915babefca46ff0a40b48bd4e60c2d7dc2fd8af","schema_version":"1.0","event_id":"sha256:39a5dac9a7cfe8320a7886977915babefca46ff0a40b48bd4e60c2d7dc2fd8af"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GAI2Y34JJOWP5UZIRSHLOYVQP4/bundle.json","state_url":"https://pith.science/pith/GAI2Y34JJOWP5UZIRSHLOYVQP4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GAI2Y34JJOWP5UZIRSHLOYVQP4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-25T04:23:05Z","links":{"resolver":"https://pith.science/pith/GAI2Y34JJOWP5UZIRSHLOYVQP4","bundle":"https://pith.science/pith/GAI2Y34JJOWP5UZIRSHLOYVQP4/bundle.json","state":"https://pith.science/pith/GAI2Y34JJOWP5UZIRSHLOYVQP4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GAI2Y34JJOWP5UZIRSHLOYVQP4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:GAI2Y34JJOWP5UZIRSHLOYVQP4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"80eed141fbe4864c76331ed1f989053b85c064f97cf85d7b3c8974643241fdeb","cross_cats_sorted":["cs.LG","physics.soc-ph"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-01-28T17:41:19Z","title_canon_sha256":"c608dd9a503bb1a15a51dd7a425568984ee8811a1b95ae0240105baf9263e8af"},"schema_version":"1.0","source":{"id":"1901.09848","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1901.09848","created_at":"2026-05-17T23:55:23Z"},{"alias_kind":"arxiv_version","alias_value":"1901.09848v1","created_at":"2026-05-17T23:55:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1901.09848","created_at":"2026-05-17T23:55:23Z"},{"alias_kind":"pith_short_12","alias_value":"GAI2Y34JJOWP","created_at":"2026-05-18T12:33:18Z"},{"alias_kind":"pith_short_16","alias_value":"GAI2Y34JJOWP5UZI","created_at":"2026-05-18T12:33:18Z"},{"alias_kind":"pith_short_8","alias_value":"GAI2Y34J","created_at":"2026-05-18T12:33:18Z"}],"graph_snapshots":[{"event_id":"sha256:39a5dac9a7cfe8320a7886977915babefca46ff0a40b48bd4e60c2d7dc2fd8af","target":"graph","created_at":"2026-05-17T23:55:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Topic models are in widespread use in natural language processing and beyond. Here, we propose a new framework for the evaluation of probabilistic topic modeling algorithms based on synthetic corpora containing an unambiguously defined ground truth topic structure. The major innovation of our approach is the ability to quantify the agreement between the planted and inferred topic structures by comparing the assigned topic labels at the level of the tokens. In experiments, our approach yields novel insights about the relative strengths of topic models as corpus characteristics vary, and the fir","authors_text":"Doug Downey, Hanyu Shi, Isabel Diersen, Luis A. N. Amaral, Martin Gerlach","cross_cats":["cs.LG","physics.soc-ph"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-01-28T17:41:19Z","title":"A new evaluation framework for topic modeling algorithms based on synthetic corpora"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1901.09848","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3349c54694cdb2562c78b93e215650ec48320b7ef2837944036f64fba4e880cc","target":"record","created_at":"2026-05-17T23:55:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"80eed141fbe4864c76331ed1f989053b85c064f97cf85d7b3c8974643241fdeb","cross_cats_sorted":["cs.LG","physics.soc-ph"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-01-28T17:41:19Z","title_canon_sha256":"c608dd9a503bb1a15a51dd7a425568984ee8811a1b95ae0240105baf9263e8af"},"schema_version":"1.0","source":{"id":"1901.09848","kind":"arxiv","version":1}},"canonical_sha256":"3011ac6f894bacfed3288c8eb762b07f265aa5e0aa7c540958f2bb29a1be15d2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3011ac6f894bacfed3288c8eb762b07f265aa5e0aa7c540958f2bb29a1be15d2","first_computed_at":"2026-05-17T23:55:23.674253Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-17T23:55:23.674253Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"qocYFJhdN+KoEOWOlEkd9ZFiRXm2fD+w/6a2VzOmrxvEocxs09mu3g9xaQB+jrZdKsXcfqxrVSVale/0GSqyDg==","signature_status":"signed_v1","signed_at":"2026-05-17T23:55:23.674726Z","signed_message":"canonical_sha256_bytes"},"source_id":"1901.09848","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3349c54694cdb2562c78b93e215650ec48320b7ef2837944036f64fba4e880cc","sha256:39a5dac9a7cfe8320a7886977915babefca46ff0a40b48bd4e60c2d7dc2fd8af"],"state_sha256":"ab77e55db7487e6736fc5941fc9ab76a703fa791d28c84b0d1521a1b4365b4c9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ol0/uVknjncu9i0+HW1tnhrJJc4p6rzFFllJYZj2LWOen2A8x2AS/joWv/zcw09SzJwKACEDxRGaexbKqBYeDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-25T04:23:05.760275Z","bundle_sha256":"dad8194951d7f419ff3dc7ab73f58e66a21c68c3e4834ddd7fe5c7dcde35a13b"}}