{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZDOGVWXIS532KJY7AY25XN3EFI","short_pith_number":"pith:ZDOGVWXI","schema_version":"1.0","canonical_sha256":"c8dc6adae89777a5271f0635dbb7642a2487464170de365185522cadbf513833","source":{"kind":"arxiv","id":"2306.05644","version":2},"attestation_state":"computed","paper":{"title":"WSPAlign: Word Alignment Pre-training via Large-Scale Weakly Supervised Span Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masaaki Nagata, Qiyu Wu, Yoshimasa Tsuruoka","submitted_at":"2023-06-09T03:11:42Z","abstract_excerpt":"Most existing word alignment methods rely on manual alignment datasets or parallel corpora, which limits their usefulness. Here, to mitigate the dependence on manual data, we broaden the source of supervision by relaxing the requirement for correct, fully-aligned, and parallel sentences. Specifically, we make noisy, partially aligned, and non-parallel paragraphs. We then use such a large-scale weakly-supervised dataset for word alignment pre-training via span prediction. Extensive experiments with various settings empirically demonstrate that our approach, which is named WSPAlign, is an effect"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.05644","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-09T03:11:42Z","cross_cats_sorted":[],"title_canon_sha256":"8223dc1514aae3d0db8a14ac5d09721e718ecf1fa4257db26e78852c3773e950","abstract_canon_sha256":"f070374bc2fa75e0ac71d077a62de394c1371e899b4ece9f79152ee71facc95b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:30.075717Z","signature_b64":"R/YQmzzxhCRg4yfAHYFConskzhmm+RIK0aAEu7HgDObQbv0glLtr9V9llg0UVPvawYbZo6QKf131wvkNgOAFAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c8dc6adae89777a5271f0635dbb7642a2487464170de365185522cadbf513833","last_reissued_at":"2026-07-05T07:02:30.075306Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:30.075306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WSPAlign: Word Alignment Pre-training via Large-Scale Weakly Supervised Span Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masaaki Nagata, Qiyu Wu, Yoshimasa Tsuruoka","submitted_at":"2023-06-09T03:11:42Z","abstract_excerpt":"Most existing word alignment methods rely on manual alignment datasets or parallel corpora, which limits their usefulness. Here, to mitigate the dependence on manual data, we broaden the source of supervision by relaxing the requirement for correct, fully-aligned, and parallel sentences. Specifically, we make noisy, partially aligned, and non-parallel paragraphs. We then use such a large-scale weakly-supervised dataset for word alignment pre-training via span prediction. Extensive experiments with various settings empirically demonstrate that our approach, which is named WSPAlign, is an effect"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.05644","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.05644/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.05644","created_at":"2026-07-05T07:02:30.075376+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.05644v2","created_at":"2026-07-05T07:02:30.075376+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.05644","created_at":"2026-07-05T07:02:30.075376+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZDOGVWXIS532","created_at":"2026-07-05T07:02:30.075376+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZDOGVWXIS532KJY7","created_at":"2026-07-05T07:02:30.075376+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZDOGVWXI","created_at":"2026-07-05T07:02:30.075376+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI","json":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI.json","graph_json":"https://pith.science/api/pith-number/ZDOGVWXIS532KJY7AY25XN3EFI/graph.json","events_json":"https://pith.science/api/pith-number/ZDOGVWXIS532KJY7AY25XN3EFI/events.json","paper":"https://pith.science/paper/ZDOGVWXI"},"agent_actions":{"view_html":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI","download_json":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI.json","view_paper":"https://pith.science/paper/ZDOGVWXI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.05644&json=true","fetch_graph":"https://pith.science/api/pith-number/ZDOGVWXIS532KJY7AY25XN3EFI/graph.json","fetch_events":"https://pith.science/api/pith-number/ZDOGVWXIS532KJY7AY25XN3EFI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI/action/storage_attestation","attest_author":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI/action/author_attestation","sign_citation":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI/action/citation_signature","submit_replication":"https://pith.science/pith/ZDOGVWXIS532KJY7AY25XN3EFI/action/replication_record"}},"created_at":"2026-07-05T07:02:30.075376+00:00","updated_at":"2026-07-05T07:02:30.075376+00:00"}