{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:O6WWTS4HT6EKWB7O24TUNNQEQW","short_pith_number":"pith:O6WWTS4H","schema_version":"1.0","canonical_sha256":"77ad69cb879f88ab07eed72746b60485a93c6b10e6163f9fbe98c207cea518e5","source":{"kind":"arxiv","id":"2107.03684","version":1},"attestation_state":"computed","paper":{"title":"Assigning Topics to Documents by Successive Projections","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.TH"],"primary_cat":"math.ST","authors_text":"Alexandre Tsybakov (CREST), Maxim Panov (Skoltech), Olga Klopp (CREST), Suzanne Sigalla (CREST)","submitted_at":"2021-07-08T08:58:35Z","abstract_excerpt":"Topic models provide a useful tool to organize and understand the structure of large corpora of text documents, in particular, to discover hidden thematic structure. Clustering documents from big unstructured corpora into topics is an important task in various areas, such as image analysis, e-commerce, social networks, population genetics. A common approach to topic modeling is to associate each topic with a probability distribution on the dictionary of words and to consider each document as a mixture of topics. Since the number of topics is typically substantially smaller than the size of the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.03684","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.ST","submitted_at":"2021-07-08T08:58:35Z","cross_cats_sorted":["stat.TH"],"title_canon_sha256":"8f4846e7b58a2e945cc13e5833534c4abbd69e2e30df1453833f19f7e5db8f39","abstract_canon_sha256":"54c1f3d3bfdbfc70e0d9b0ce2ef03f3cb0c29f8121fc52b32dbdb7950fb23b92"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:56:18.441637Z","signature_b64":"f7EP3d2nAd30GlKg80Vb5tct0xBHkX7j03Kx0yiWy75AGZpUxLHSr/MI1Z6Dx7m5p8Xu7LKYbiqA5S3Sgp7qBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77ad69cb879f88ab07eed72746b60485a93c6b10e6163f9fbe98c207cea518e5","last_reissued_at":"2026-07-05T02:56:18.441276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:56:18.441276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assigning Topics to Documents by Successive Projections","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.TH"],"primary_cat":"math.ST","authors_text":"Alexandre Tsybakov (CREST), Maxim Panov (Skoltech), Olga Klopp (CREST), Suzanne Sigalla (CREST)","submitted_at":"2021-07-08T08:58:35Z","abstract_excerpt":"Topic models provide a useful tool to organize and understand the structure of large corpora of text documents, in particular, to discover hidden thematic structure. Clustering documents from big unstructured corpora into topics is an important task in various areas, such as image analysis, e-commerce, social networks, population genetics. A common approach to topic modeling is to associate each topic with a probability distribution on the dictionary of words and to consider each document as a mixture of topics. Since the number of topics is typically substantially smaller than the size of the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.03684","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.03684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.03684","created_at":"2026-07-05T02:56:18.441337+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.03684v1","created_at":"2026-07-05T02:56:18.441337+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.03684","created_at":"2026-07-05T02:56:18.441337+00:00"},{"alias_kind":"pith_short_12","alias_value":"O6WWTS4HT6EK","created_at":"2026-07-05T02:56:18.441337+00:00"},{"alias_kind":"pith_short_16","alias_value":"O6WWTS4HT6EKWB7O","created_at":"2026-07-05T02:56:18.441337+00:00"},{"alias_kind":"pith_short_8","alias_value":"O6WWTS4H","created_at":"2026-07-05T02:56:18.441337+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW","json":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW.json","graph_json":"https://pith.science/api/pith-number/O6WWTS4HT6EKWB7O24TUNNQEQW/graph.json","events_json":"https://pith.science/api/pith-number/O6WWTS4HT6EKWB7O24TUNNQEQW/events.json","paper":"https://pith.science/paper/O6WWTS4H"},"agent_actions":{"view_html":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW","download_json":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW.json","view_paper":"https://pith.science/paper/O6WWTS4H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.03684&json=true","fetch_graph":"https://pith.science/api/pith-number/O6WWTS4HT6EKWB7O24TUNNQEQW/graph.json","fetch_events":"https://pith.science/api/pith-number/O6WWTS4HT6EKWB7O24TUNNQEQW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW/action/storage_attestation","attest_author":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW/action/author_attestation","sign_citation":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW/action/citation_signature","submit_replication":"https://pith.science/pith/O6WWTS4HT6EKWB7O24TUNNQEQW/action/replication_record"}},"created_at":"2026-07-05T02:56:18.441337+00:00","updated_at":"2026-07-05T02:56:18.441337+00:00"}