{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:FOPO24RYP6B3XJNSVEIEYRSQMQ","short_pith_number":"pith:FOPO24RY","schema_version":"1.0","canonical_sha256":"2b9eed72387f83bba5b2a9104c46506418b591a5e09d97eca298dbebba350147","source":{"kind":"arxiv","id":"2205.04651","version":1},"attestation_state":"computed","paper":{"title":"ParaCotta: Synthetic Multilingual Paraphrase Corpora from the Most Diverse Translation Sample Pair","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Mahendra Data, Nadhifa Zulfa, Philip Arthur, Radityo Eko Prasojo, Salma Qonitah, Suci Fitriany, Tirana Noor Fatyanosa, Tomi Santoso","submitted_at":"2022-05-10T03:40:14Z","abstract_excerpt":"We release our synthetic parallel paraphrase corpus across 17 languages: Arabic, Catalan, Czech, German, English, Spanish, Estonian, French, Hindi, Indonesian, Italian, Dutch, Romanian, Russian, Swedish, Vietnamese, and Chinese. Our method relies only on monolingual data and a neural machine translation system to generate paraphrases, hence simple to apply. We generate multiple translation samples using beam search and choose the most lexically diverse pair according to their sentence BLEU. We compare our generated corpus with the \\texttt{ParaBank2}. According to our evaluation, our synthetic "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.04651","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-05-10T03:40:14Z","cross_cats_sorted":[],"title_canon_sha256":"bb274784174f63696608595a17e5b5036a4c7682258d8f57e1745573237745ac","abstract_canon_sha256":"39a70eaf0166720e40d159589b80cf1c05b0f57f0ffed4ef618888d83515cfff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:21:38.517019Z","signature_b64":"l1ByCKoRg/Qe/wrkkiuvaqvRLdvqOE5HXvk4CyVZXW43mowu6KR6ZAsYPpuWbsLLCOXgzwPJDDK4OWLJKfjIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b9eed72387f83bba5b2a9104c46506418b591a5e09d97eca298dbebba350147","last_reissued_at":"2026-07-05T04:21:38.516577Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:21:38.516577Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ParaCotta: Synthetic Multilingual Paraphrase Corpora from the Most Diverse Translation Sample Pair","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Mahendra Data, Nadhifa Zulfa, Philip Arthur, Radityo Eko Prasojo, Salma Qonitah, Suci Fitriany, Tirana Noor Fatyanosa, Tomi Santoso","submitted_at":"2022-05-10T03:40:14Z","abstract_excerpt":"We release our synthetic parallel paraphrase corpus across 17 languages: Arabic, Catalan, Czech, German, English, Spanish, Estonian, French, Hindi, Indonesian, Italian, Dutch, Romanian, Russian, Swedish, Vietnamese, and Chinese. Our method relies only on monolingual data and a neural machine translation system to generate paraphrases, hence simple to apply. We generate multiple translation samples using beam search and choose the most lexically diverse pair according to their sentence BLEU. We compare our generated corpus with the \\texttt{ParaBank2}. According to our evaluation, our synthetic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.04651","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.04651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.04651","created_at":"2026-07-05T04:21:38.516632+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.04651v1","created_at":"2026-07-05T04:21:38.516632+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.04651","created_at":"2026-07-05T04:21:38.516632+00:00"},{"alias_kind":"pith_short_12","alias_value":"FOPO24RYP6B3","created_at":"2026-07-05T04:21:38.516632+00:00"},{"alias_kind":"pith_short_16","alias_value":"FOPO24RYP6B3XJNS","created_at":"2026-07-05T04:21:38.516632+00:00"},{"alias_kind":"pith_short_8","alias_value":"FOPO24RY","created_at":"2026-07-05T04:21:38.516632+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.17444","citing_title":"MahaParaphrase: A Marathi Paraphrase Detection Corpus and BERT-based Models","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ","json":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ.json","graph_json":"https://pith.science/api/pith-number/FOPO24RYP6B3XJNSVEIEYRSQMQ/graph.json","events_json":"https://pith.science/api/pith-number/FOPO24RYP6B3XJNSVEIEYRSQMQ/events.json","paper":"https://pith.science/paper/FOPO24RY"},"agent_actions":{"view_html":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ","download_json":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ.json","view_paper":"https://pith.science/paper/FOPO24RY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.04651&json=true","fetch_graph":"https://pith.science/api/pith-number/FOPO24RYP6B3XJNSVEIEYRSQMQ/graph.json","fetch_events":"https://pith.science/api/pith-number/FOPO24RYP6B3XJNSVEIEYRSQMQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ/action/storage_attestation","attest_author":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ/action/author_attestation","sign_citation":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ/action/citation_signature","submit_replication":"https://pith.science/pith/FOPO24RYP6B3XJNSVEIEYRSQMQ/action/replication_record"}},"created_at":"2026-07-05T04:21:38.516632+00:00","updated_at":"2026-07-05T04:21:38.516632+00:00"}