{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:KGAETT5QFQUC5YZ7FRJPIKOQTS","short_pith_number":"pith:KGAETT5Q","schema_version":"1.0","canonical_sha256":"518049cfb02c282ee33f2c52f429d09c95aaba22da17cc019927094763254029","source":{"kind":"arxiv","id":"1912.05200","version":2},"attestation_state":"computed","paper":{"title":"Automatic Spanish Translation of the SQuAD Dataset for Multilingual Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Casimiro Pio Carrino, Jos\\'e A. R. Fonollosa, Marta R. Costa-juss\\`a","submitted_at":"2019-12-11T09:33:21Z","abstract_excerpt":"Recently, multilingual question answering became a crucial research topic, and it is receiving increased interest in the NLP community. However, the unavailability of large-scale datasets makes it challenging to train multilingual QA systems with performance comparable to the English ones. In this work, we develop the Translate Align Retrieve (TAR) method to automatically translate the Stanford Question Answering Dataset (SQuAD) v1.1 to Spanish. We then used this dataset to train Spanish QA systems by fine-tuning a Multilingual-BERT model. Finally, we evaluated our QA models with the recently "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.05200","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2019-12-11T09:33:21Z","cross_cats_sorted":[],"title_canon_sha256":"f1ddfd0d5dbfd6658a10cd2cf34a707bfe600d646f5040ecbb1057f777f6f8b4","abstract_canon_sha256":"25b10b9c738a850494f185ba25fc692fe1169a2f6c625f42a2ce746813f1bb9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:25:50.646291Z","signature_b64":"+jNUyDG7UBXU2Ei9GUX+37vVkuZdHlmV5VvoAZGT+Mogt5MjBl+npDFoTyMxfVB+OICICiPFrTib1lRy5fnaAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"518049cfb02c282ee33f2c52f429d09c95aaba22da17cc019927094763254029","last_reissued_at":"2026-07-05T00:25:50.645746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:25:50.645746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automatic Spanish Translation of the SQuAD Dataset for Multilingual Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Casimiro Pio Carrino, Jos\\'e A. R. Fonollosa, Marta R. Costa-juss\\`a","submitted_at":"2019-12-11T09:33:21Z","abstract_excerpt":"Recently, multilingual question answering became a crucial research topic, and it is receiving increased interest in the NLP community. However, the unavailability of large-scale datasets makes it challenging to train multilingual QA systems with performance comparable to the English ones. In this work, we develop the Translate Align Retrieve (TAR) method to automatically translate the Stanford Question Answering Dataset (SQuAD) v1.1 to Spanish. We then used this dataset to train Spanish QA systems by fine-tuning a Multilingual-BERT model. Finally, we evaluated our QA models with the recently "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.05200","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.05200/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.05200","created_at":"2026-07-05T00:25:50.645813+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.05200v2","created_at":"2026-07-05T00:25:50.645813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.05200","created_at":"2026-07-05T00:25:50.645813+00:00"},{"alias_kind":"pith_short_12","alias_value":"KGAETT5QFQUC","created_at":"2026-07-05T00:25:50.645813+00:00"},{"alias_kind":"pith_short_16","alias_value":"KGAETT5QFQUC5YZ7","created_at":"2026-07-05T00:25:50.645813+00:00"},{"alias_kind":"pith_short_8","alias_value":"KGAETT5Q","created_at":"2026-07-05T00:25:50.645813+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02047","citing_title":"AmaSQuAD: A Benchmark for Amharic Extractive Question Answering","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS","json":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS.json","graph_json":"https://pith.science/api/pith-number/KGAETT5QFQUC5YZ7FRJPIKOQTS/graph.json","events_json":"https://pith.science/api/pith-number/KGAETT5QFQUC5YZ7FRJPIKOQTS/events.json","paper":"https://pith.science/paper/KGAETT5Q"},"agent_actions":{"view_html":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS","download_json":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS.json","view_paper":"https://pith.science/paper/KGAETT5Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.05200&json=true","fetch_graph":"https://pith.science/api/pith-number/KGAETT5QFQUC5YZ7FRJPIKOQTS/graph.json","fetch_events":"https://pith.science/api/pith-number/KGAETT5QFQUC5YZ7FRJPIKOQTS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS/action/storage_attestation","attest_author":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS/action/author_attestation","sign_citation":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS/action/citation_signature","submit_replication":"https://pith.science/pith/KGAETT5QFQUC5YZ7FRJPIKOQTS/action/replication_record"}},"created_at":"2026-07-05T00:25:50.645813+00:00","updated_at":"2026-07-05T00:25:50.645813+00:00"}