{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:RNWUP6NPNJAIQ6CK43JI62PCEC","short_pith_number":"pith:RNWUP6NP","schema_version":"1.0","canonical_sha256":"8b6d47f9af6a4088784ae6d28f69e22091a86ae7a88871d3a712cf05f626ef94","source":{"kind":"arxiv","id":"2202.02398","version":1},"attestation_state":"computed","paper":{"title":"Pir\\'a: A Bilingual Portuguese-English Dataset for Question-Answering about the Ocean","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anarosa A. F. Brand\\~ao, Andr\\'e F. A. Paschoal, Andr\\'e S. Oliveira, Anna H. R. Costa, Fabio G. Cozman, Fl\\'avio Nakasato, Karina V. Delgado, Marcos M. Jos\\'e, Paulo Pirozelli, Sarajane M. Peres, Valdinei Freire","submitted_at":"2022-02-04T21:29:45Z","abstract_excerpt":"Current research in natural language processing is highly dependent on carefully produced corpora. Most existing resources focus on English; some resources focus on languages such as Chinese and French; few resources deal with more than one language. This paper presents the Pir\\'a dataset, a large set of questions and answers about the ocean and the Brazilian coast both in Portuguese and English. Pir\\'a is, to the best of our knowledge, the first QA dataset with supporting texts in Portuguese, and, perhaps more importantly, the first bilingual QA dataset that includes this language. The Pir\\'a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.02398","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-04T21:29:45Z","cross_cats_sorted":[],"title_canon_sha256":"77faad923ddf71fbcd25515541cd1b757812cc43404b9f6ac321753d901ec972","abstract_canon_sha256":"22c2246546a08765af26b305c7d1df4e376e65b17e23529ef1f16973a23e1fb0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:54:28.471687Z","signature_b64":"hzqWL85H2KKSm5+YXqz1kZ4M0gDhpKUPC1e/HN3OqI4NqcPn+EE7b/fLDZGj9/GwDZ0W1v5TtGwLrVI009JYAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b6d47f9af6a4088784ae6d28f69e22091a86ae7a88871d3a712cf05f626ef94","last_reissued_at":"2026-07-05T03:54:28.471331Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:54:28.471331Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pir\\'a: A Bilingual Portuguese-English Dataset for Question-Answering about the Ocean","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anarosa A. F. Brand\\~ao, Andr\\'e F. A. Paschoal, Andr\\'e S. Oliveira, Anna H. R. Costa, Fabio G. Cozman, Fl\\'avio Nakasato, Karina V. Delgado, Marcos M. Jos\\'e, Paulo Pirozelli, Sarajane M. Peres, Valdinei Freire","submitted_at":"2022-02-04T21:29:45Z","abstract_excerpt":"Current research in natural language processing is highly dependent on carefully produced corpora. Most existing resources focus on English; some resources focus on languages such as Chinese and French; few resources deal with more than one language. This paper presents the Pir\\'a dataset, a large set of questions and answers about the ocean and the Brazilian coast both in Portuguese and English. Pir\\'a is, to the best of our knowledge, the first QA dataset with supporting texts in Portuguese, and, perhaps more importantly, the first bilingual QA dataset that includes this language. The Pir\\'a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.02398","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.02398/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.02398","created_at":"2026-07-05T03:54:28.471387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.02398v1","created_at":"2026-07-05T03:54:28.471387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.02398","created_at":"2026-07-05T03:54:28.471387+00:00"},{"alias_kind":"pith_short_12","alias_value":"RNWUP6NPNJAI","created_at":"2026-07-05T03:54:28.471387+00:00"},{"alias_kind":"pith_short_16","alias_value":"RNWUP6NPNJAIQ6CK","created_at":"2026-07-05T03:54:28.471387+00:00"},{"alias_kind":"pith_short_8","alias_value":"RNWUP6NP","created_at":"2026-07-05T03:54:28.471387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20691","citing_title":"Specific Domain Ontology Construction Using Large Language Models","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC","json":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC.json","graph_json":"https://pith.science/api/pith-number/RNWUP6NPNJAIQ6CK43JI62PCEC/graph.json","events_json":"https://pith.science/api/pith-number/RNWUP6NPNJAIQ6CK43JI62PCEC/events.json","paper":"https://pith.science/paper/RNWUP6NP"},"agent_actions":{"view_html":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC","download_json":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC.json","view_paper":"https://pith.science/paper/RNWUP6NP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.02398&json=true","fetch_graph":"https://pith.science/api/pith-number/RNWUP6NPNJAIQ6CK43JI62PCEC/graph.json","fetch_events":"https://pith.science/api/pith-number/RNWUP6NPNJAIQ6CK43JI62PCEC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC/action/storage_attestation","attest_author":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC/action/author_attestation","sign_citation":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC/action/citation_signature","submit_replication":"https://pith.science/pith/RNWUP6NPNJAIQ6CK43JI62PCEC/action/replication_record"}},"created_at":"2026-07-05T03:54:28.471387+00:00","updated_at":"2026-07-05T03:54:28.471387+00:00"}