{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:CTSXQA4SN2WGG7FSSV4XOQQYCY","short_pith_number":"pith:CTSXQA4S","schema_version":"1.0","canonical_sha256":"14e57803926eac637cb2957977421816048f1d334a1b89538bf285546fcb906d","source":{"kind":"arxiv","id":"2111.02362","version":1},"attestation_state":"computed","paper":{"title":"HmBlogs: A big general Persian corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hamzeh Motahari Khansari, Mehrnoush Shamsfard","submitted_at":"2021-11-03T17:26:52Z","abstract_excerpt":"This paper introduces the hmBlogs corpus for Persian, as a low resource language. This corpus has been prepared based on a collection of nearly 20 million blog posts over a period of about 15 years from a space of Persian blogs and includes more than 6.8 billion tokens. It can be claimed that this corpus is currently the largest Persian corpus that has been prepared independently for the Persian language. This corpus is presented in both raw and preprocessed forms, and based on the preprocessed corpus some word embedding models are produced. By the provided models, the hmBlogs is compared with"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.02362","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-11-03T17:26:52Z","cross_cats_sorted":[],"title_canon_sha256":"dccf532a136003c26ba5fce50593d6028d2bc99c4bf0c5c8d7d17fed42fd8ad1","abstract_canon_sha256":"2f85bfe71a1e1b5dca1220a518a9c330da92ca292a2038c22ece85fe9c50ba86"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:28:49.923657Z","signature_b64":"eEWe1UyjS3wAs/kOkrWntm1A8gF+uD82a0WyCbl1uUjEi3HFyefEJ+Tz5kYvLjlSRNI+friVCQAGF9cVjfsqDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14e57803926eac637cb2957977421816048f1d334a1b89538bf285546fcb906d","last_reissued_at":"2026-07-05T03:28:49.923091Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:28:49.923091Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HmBlogs: A big general Persian corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hamzeh Motahari Khansari, Mehrnoush Shamsfard","submitted_at":"2021-11-03T17:26:52Z","abstract_excerpt":"This paper introduces the hmBlogs corpus for Persian, as a low resource language. This corpus has been prepared based on a collection of nearly 20 million blog posts over a period of about 15 years from a space of Persian blogs and includes more than 6.8 billion tokens. It can be claimed that this corpus is currently the largest Persian corpus that has been prepared independently for the Persian language. This corpus is presented in both raw and preprocessed forms, and based on the preprocessed corpus some word embedding models are produced. By the provided models, the hmBlogs is compared with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.02362","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.02362/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.02362","created_at":"2026-07-05T03:28:49.923171+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.02362v1","created_at":"2026-07-05T03:28:49.923171+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.02362","created_at":"2026-07-05T03:28:49.923171+00:00"},{"alias_kind":"pith_short_12","alias_value":"CTSXQA4SN2WG","created_at":"2026-07-05T03:28:49.923171+00:00"},{"alias_kind":"pith_short_16","alias_value":"CTSXQA4SN2WGG7FS","created_at":"2026-07-05T03:28:49.923171+00:00"},{"alias_kind":"pith_short_8","alias_value":"CTSXQA4S","created_at":"2026-07-05T03:28:49.923171+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.14690","citing_title":"FarsEval-PKBETS: A new diverse benchmark for evaluating Persian large language models","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY","json":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY.json","graph_json":"https://pith.science/api/pith-number/CTSXQA4SN2WGG7FSSV4XOQQYCY/graph.json","events_json":"https://pith.science/api/pith-number/CTSXQA4SN2WGG7FSSV4XOQQYCY/events.json","paper":"https://pith.science/paper/CTSXQA4S"},"agent_actions":{"view_html":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY","download_json":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY.json","view_paper":"https://pith.science/paper/CTSXQA4S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.02362&json=true","fetch_graph":"https://pith.science/api/pith-number/CTSXQA4SN2WGG7FSSV4XOQQYCY/graph.json","fetch_events":"https://pith.science/api/pith-number/CTSXQA4SN2WGG7FSSV4XOQQYCY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY/action/storage_attestation","attest_author":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY/action/author_attestation","sign_citation":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY/action/citation_signature","submit_replication":"https://pith.science/pith/CTSXQA4SN2WGG7FSSV4XOQQYCY/action/replication_record"}},"created_at":"2026-07-05T03:28:49.923171+00:00","updated_at":"2026-07-05T03:28:49.923171+00:00"}