{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:TNT7WQYGSNE3BB7QZUEEYJ5FVJ","short_pith_number":"pith:TNT7WQYG","schema_version":"1.0","canonical_sha256":"9b67fb43069349b087f0cd084c27a5aa5bebd03f8db5052dc153a59267541e7a","source":{"kind":"arxiv","id":"2210.14763","version":1},"attestation_state":"computed","paper":{"title":"ProSiT! Latent Variable Discovery with PROgressive SImilarity Thresholds","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dirk Hovy, Federico Bianchi, Tommaso Fornaciari","submitted_at":"2022-10-26T14:52:44Z","abstract_excerpt":"The most common ways to explore latent document dimensions are topic models and clustering methods. However, topic models have several drawbacks: e.g., they require us to choose the number of latent dimensions a priori, and the results are stochastic. Most clustering methods have the same issues and lack flexibility in various ways, such as not accounting for the influence of different topics on single documents, forcing word-descriptors to belong to a single topic (hard-clustering) or necessarily relying on word representations. We propose PROgressive SImilarity Thresholds - ProSiT, a determi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.14763","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-10-26T14:52:44Z","cross_cats_sorted":[],"title_canon_sha256":"11da21aab9816da263d89fabcd39137fb2fd8ccd72d88e1b068a1e4413406364","abstract_canon_sha256":"2649f4fadae2719290b117ac0103735b2e6f506bdb59298e1c1a54823b5186f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:10:49.504936Z","signature_b64":"dA2pDVQlhY2j9YvLkTa7DLFVyZIZeEBbVmhroBjwgVPWQZuJNCLzz12RdDtnAtBBGnb8qUZbioOnikxc9yaoBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b67fb43069349b087f0cd084c27a5aa5bebd03f8db5052dc153a59267541e7a","last_reissued_at":"2026-07-05T05:10:49.504557Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:10:49.504557Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ProSiT! Latent Variable Discovery with PROgressive SImilarity Thresholds","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dirk Hovy, Federico Bianchi, Tommaso Fornaciari","submitted_at":"2022-10-26T14:52:44Z","abstract_excerpt":"The most common ways to explore latent document dimensions are topic models and clustering methods. However, topic models have several drawbacks: e.g., they require us to choose the number of latent dimensions a priori, and the results are stochastic. Most clustering methods have the same issues and lack flexibility in various ways, such as not accounting for the influence of different topics on single documents, forcing word-descriptors to belong to a single topic (hard-clustering) or necessarily relying on word representations. We propose PROgressive SImilarity Thresholds - ProSiT, a determi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.14763","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.14763/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.14763","created_at":"2026-07-05T05:10:49.504613+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.14763v1","created_at":"2026-07-05T05:10:49.504613+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.14763","created_at":"2026-07-05T05:10:49.504613+00:00"},{"alias_kind":"pith_short_12","alias_value":"TNT7WQYGSNE3","created_at":"2026-07-05T05:10:49.504613+00:00"},{"alias_kind":"pith_short_16","alias_value":"TNT7WQYGSNE3BB7Q","created_at":"2026-07-05T05:10:49.504613+00:00"},{"alias_kind":"pith_short_8","alias_value":"TNT7WQYG","created_at":"2026-07-05T05:10:49.504613+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ","json":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ.json","graph_json":"https://pith.science/api/pith-number/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/graph.json","events_json":"https://pith.science/api/pith-number/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/events.json","paper":"https://pith.science/paper/TNT7WQYG"},"agent_actions":{"view_html":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ","download_json":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ.json","view_paper":"https://pith.science/paper/TNT7WQYG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.14763&json=true","fetch_graph":"https://pith.science/api/pith-number/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/graph.json","fetch_events":"https://pith.science/api/pith-number/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/action/storage_attestation","attest_author":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/action/author_attestation","sign_citation":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/action/citation_signature","submit_replication":"https://pith.science/pith/TNT7WQYGSNE3BB7QZUEEYJ5FVJ/action/replication_record"}},"created_at":"2026-07-05T05:10:49.504613+00:00","updated_at":"2026-07-05T05:10:49.504613+00:00"}