{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MHDMWEVHO5ENI6EC3GHVIMCVZG","short_pith_number":"pith:MHDMWEVH","schema_version":"1.0","canonical_sha256":"61c6cb12a77748d47882d98f543055c9ac43eda85d726be67e5e8fa7c26b733a","source":{"kind":"arxiv","id":"2310.06730","version":1},"attestation_state":"computed","paper":{"title":"Sparse topic modeling via spectral decomposition and thresholding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"stat.ME","authors_text":"Claire Donnat, Huy Tran, Yating Liu","submitted_at":"2023-10-10T15:54:20Z","abstract_excerpt":"The probabilistic Latent Semantic Indexing model assumes that the expectation of the corpus matrix is low-rank and can be written as the product of a topic-word matrix and a word-document matrix. In this paper, we study the estimation of the topic-word matrix under the additional assumption that the ordered entries of its columns rapidly decay to zero. This sparsity assumption is motivated by the empirical observation that the word frequencies in a text often adhere to Zipf's law. We introduce a new spectral procedure for estimating the topic-word matrix that thresholds words based on their co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06730","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ME","submitted_at":"2023-10-10T15:54:20Z","cross_cats_sorted":[],"title_canon_sha256":"064acbd4a1f9ba4698142a4d922b49ab56094c7c8057f9076f50f1f764e0ed92","abstract_canon_sha256":"34d5f7c36e6ed43c76e86a2a3f326b8af18e41335f163110362ed83a0900c922"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:15.306120Z","signature_b64":"LjxhGM2CDtR0E08eYai2penlBst4gbwsnNu/ad/NoqRaRLzcZzhQScwyRJNMFGy9irAYDimQjNimdacMrTKUAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61c6cb12a77748d47882d98f543055c9ac43eda85d726be67e5e8fa7c26b733a","last_reissued_at":"2026-07-05T06:59:15.305645Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:15.305645Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparse topic modeling via spectral decomposition and thresholding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"stat.ME","authors_text":"Claire Donnat, Huy Tran, Yating Liu","submitted_at":"2023-10-10T15:54:20Z","abstract_excerpt":"The probabilistic Latent Semantic Indexing model assumes that the expectation of the corpus matrix is low-rank and can be written as the product of a topic-word matrix and a word-document matrix. In this paper, we study the estimation of the topic-word matrix under the additional assumption that the ordered entries of its columns rapidly decay to zero. This sparsity assumption is motivated by the empirical observation that the word frequencies in a text often adhere to Zipf's law. We introduce a new spectral procedure for estimating the topic-word matrix that thresholds words based on their co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06730","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06730/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06730","created_at":"2026-07-05T06:59:15.305703+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06730v1","created_at":"2026-07-05T06:59:15.305703+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06730","created_at":"2026-07-05T06:59:15.305703+00:00"},{"alias_kind":"pith_short_12","alias_value":"MHDMWEVHO5EN","created_at":"2026-07-05T06:59:15.305703+00:00"},{"alias_kind":"pith_short_16","alias_value":"MHDMWEVHO5ENI6EC","created_at":"2026-07-05T06:59:15.305703+00:00"},{"alias_kind":"pith_short_8","alias_value":"MHDMWEVH","created_at":"2026-07-05T06:59:15.305703+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.00535","citing_title":"Tensor Topic Modeling Via HOSVD","ref_index":52,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG","json":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG.json","graph_json":"https://pith.science/api/pith-number/MHDMWEVHO5ENI6EC3GHVIMCVZG/graph.json","events_json":"https://pith.science/api/pith-number/MHDMWEVHO5ENI6EC3GHVIMCVZG/events.json","paper":"https://pith.science/paper/MHDMWEVH"},"agent_actions":{"view_html":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG","download_json":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG.json","view_paper":"https://pith.science/paper/MHDMWEVH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06730&json=true","fetch_graph":"https://pith.science/api/pith-number/MHDMWEVHO5ENI6EC3GHVIMCVZG/graph.json","fetch_events":"https://pith.science/api/pith-number/MHDMWEVHO5ENI6EC3GHVIMCVZG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG/action/storage_attestation","attest_author":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG/action/author_attestation","sign_citation":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG/action/citation_signature","submit_replication":"https://pith.science/pith/MHDMWEVHO5ENI6EC3GHVIMCVZG/action/replication_record"}},"created_at":"2026-07-05T06:59:15.305703+00:00","updated_at":"2026-07-05T06:59:15.305703+00:00"}