{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:T2NW6BLO6NWDYZNKIE66MBBOMX","short_pith_number":"pith:T2NW6BLO","schema_version":"1.0","canonical_sha256":"9e9b6f056ef36c3c65aa413de6042e65d2fe1f84f07bcd52bf14edfecb55d732","source":{"kind":"arxiv","id":"2206.15147","version":2},"attestation_state":"computed","paper":{"title":"esCorpius: A Massive Spanish Crawling Corpus","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Asier Guti\\'errez-Fandi\\~no, David Griol, David P\\'erez-Fern\\'andez, Jordi Armengol-Estap\\'e, Zoraida Callejas","submitted_at":"2022-06-30T09:29:18Z","abstract_excerpt":"In the recent years, transformer-based models have lead to significant advances in language modelling for natural language processing. However, they require a vast amount of data to be (pre-)trained and there is a lack of corpora in languages other than English. Recently, several initiatives have presented multilingual datasets obtained from automatic web crawling. However, the results in Spanish present important shortcomings, as they are either too small in comparison with other languages, or present a low quality derived from sub-optimal cleaning and deduplication. In this paper, we introdu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.15147","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-06-30T09:29:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"82a658752e23bb1287392b22f4fb8ebe2bbd82ab1150679793747e4973c6d008","abstract_canon_sha256":"8847915a79047ac31197859afe91f4b4e8db7fb45057e608b189b514b5ec760c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:36:37.933022Z","signature_b64":"D+N6tsYL9ffgn3xPcqiCfezp2b6iulIUEl07y+Tgem6l3s3tMGfDrIDQrYFURpyhbzo0+ttjK6FHEVqxH6RqDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e9b6f056ef36c3c65aa413de6042e65d2fe1f84f07bcd52bf14edfecb55d732","last_reissued_at":"2026-07-05T04:36:37.932523Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:36:37.932523Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"esCorpius: A Massive Spanish Crawling Corpus","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Asier Guti\\'errez-Fandi\\~no, David Griol, David P\\'erez-Fern\\'andez, Jordi Armengol-Estap\\'e, Zoraida Callejas","submitted_at":"2022-06-30T09:29:18Z","abstract_excerpt":"In the recent years, transformer-based models have lead to significant advances in language modelling for natural language processing. However, they require a vast amount of data to be (pre-)trained and there is a lack of corpora in languages other than English. Recently, several initiatives have presented multilingual datasets obtained from automatic web crawling. However, the results in Spanish present important shortcomings, as they are either too small in comparison with other languages, or present a low quality derived from sub-optimal cleaning and deduplication. In this paper, we introdu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.15147","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.15147/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.15147","created_at":"2026-07-05T04:36:37.932584+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.15147v2","created_at":"2026-07-05T04:36:37.932584+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.15147","created_at":"2026-07-05T04:36:37.932584+00:00"},{"alias_kind":"pith_short_12","alias_value":"T2NW6BLO6NWD","created_at":"2026-07-05T04:36:37.932584+00:00"},{"alias_kind":"pith_short_16","alias_value":"T2NW6BLO6NWDYZNK","created_at":"2026-07-05T04:36:37.932584+00:00"},{"alias_kind":"pith_short_8","alias_value":"T2NW6BLO","created_at":"2026-07-05T04:36:37.932584+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.11039","citing_title":"CC-GPX: Extracting High-Quality Annotated Geospatial Data from Common Crawl","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX","json":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX.json","graph_json":"https://pith.science/api/pith-number/T2NW6BLO6NWDYZNKIE66MBBOMX/graph.json","events_json":"https://pith.science/api/pith-number/T2NW6BLO6NWDYZNKIE66MBBOMX/events.json","paper":"https://pith.science/paper/T2NW6BLO"},"agent_actions":{"view_html":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX","download_json":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX.json","view_paper":"https://pith.science/paper/T2NW6BLO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.15147&json=true","fetch_graph":"https://pith.science/api/pith-number/T2NW6BLO6NWDYZNKIE66MBBOMX/graph.json","fetch_events":"https://pith.science/api/pith-number/T2NW6BLO6NWDYZNKIE66MBBOMX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX/action/storage_attestation","attest_author":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX/action/author_attestation","sign_citation":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX/action/citation_signature","submit_replication":"https://pith.science/pith/T2NW6BLO6NWDYZNKIE66MBBOMX/action/replication_record"}},"created_at":"2026-07-05T04:36:37.932584+00:00","updated_at":"2026-07-05T04:36:37.932584+00:00"}