{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WE3ZPSVWZXWHETTLEZJW5RLWOZ","short_pith_number":"pith:WE3ZPSVW","schema_version":"1.0","canonical_sha256":"b13797cab6cdec724e6b26536ec5767652d4f13b9a0e639f4d2e06382464dd53","source":{"kind":"arxiv","id":"2210.09389","version":1},"attestation_state":"computed","paper":{"title":"Potrika: Raw and Balanced Newspaper Datasets in the Bangla Language with Eight Topics and Five Attributes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fahad AlQurashi, Istiak Ahmad, Rashid Mehmood","submitted_at":"2022-10-17T19:37:42Z","abstract_excerpt":"Knowledge is central to human and scientific developments. Natural Language Processing (NLP) allows automated analysis and creation of knowledge. Data is a crucial NLP and machine learning ingredient. The scarcity of open datasets is a well-known problem in machine and deep learning research. This is very much the case for textual NLP datasets in English and other major world languages. For the Bangla language, the situation is even more challenging and the number of large datasets for NLP research is practically nil. We hereby present Potrika, a large single-label Bangla news article textual "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.09389","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-10-17T19:37:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e2507773c67e2b1a3d474395588e6dbe817334f2a02dc53ee56b1e71134464f5","abstract_canon_sha256":"243a99dfb720010dcf077e0cb7b1debc30884da2f1ab2eb550fe6dec54fb06a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:07:24.259593Z","signature_b64":"WbasfGUMTiKx6vCjr6Yh8zvYRCKWzDZN0s+tlKdKhy0SVIyV9vaf1hMtVpwXjEZ8Med3DDGVUtx8wiMISwfjCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b13797cab6cdec724e6b26536ec5767652d4f13b9a0e639f4d2e06382464dd53","last_reissued_at":"2026-07-05T05:07:24.259128Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:07:24.259128Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Potrika: Raw and Balanced Newspaper Datasets in the Bangla Language with Eight Topics and Five Attributes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fahad AlQurashi, Istiak Ahmad, Rashid Mehmood","submitted_at":"2022-10-17T19:37:42Z","abstract_excerpt":"Knowledge is central to human and scientific developments. Natural Language Processing (NLP) allows automated analysis and creation of knowledge. Data is a crucial NLP and machine learning ingredient. The scarcity of open datasets is a well-known problem in machine and deep learning research. This is very much the case for textual NLP datasets in English and other major world languages. For the Bangla language, the situation is even more challenging and the number of large datasets for NLP research is practically nil. We hereby present Potrika, a large single-label Bangla news article textual "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.09389","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.09389/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.09389","created_at":"2026-07-05T05:07:24.259186+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.09389v1","created_at":"2026-07-05T05:07:24.259186+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.09389","created_at":"2026-07-05T05:07:24.259186+00:00"},{"alias_kind":"pith_short_12","alias_value":"WE3ZPSVWZXWH","created_at":"2026-07-05T05:07:24.259186+00:00"},{"alias_kind":"pith_short_16","alias_value":"WE3ZPSVWZXWHETTL","created_at":"2026-07-05T05:07:24.259186+00:00"},{"alias_kind":"pith_short_8","alias_value":"WE3ZPSVW","created_at":"2026-07-05T05:07:24.259186+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.00592","citing_title":"GeoMoE: Divide-and-Conquer Motion Field Modeling with Mixture-of-Experts for Two-View Geometry","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ","json":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ.json","graph_json":"https://pith.science/api/pith-number/WE3ZPSVWZXWHETTLEZJW5RLWOZ/graph.json","events_json":"https://pith.science/api/pith-number/WE3ZPSVWZXWHETTLEZJW5RLWOZ/events.json","paper":"https://pith.science/paper/WE3ZPSVW"},"agent_actions":{"view_html":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ","download_json":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ.json","view_paper":"https://pith.science/paper/WE3ZPSVW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.09389&json=true","fetch_graph":"https://pith.science/api/pith-number/WE3ZPSVWZXWHETTLEZJW5RLWOZ/graph.json","fetch_events":"https://pith.science/api/pith-number/WE3ZPSVWZXWHETTLEZJW5RLWOZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ/action/storage_attestation","attest_author":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ/action/author_attestation","sign_citation":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ/action/citation_signature","submit_replication":"https://pith.science/pith/WE3ZPSVWZXWHETTLEZJW5RLWOZ/action/replication_record"}},"created_at":"2026-07-05T05:07:24.259186+00:00","updated_at":"2026-07-05T05:07:24.259186+00:00"}