{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:EDHGHXC6S7UAEYKM2JUYWJECKN","short_pith_number":"pith:EDHGHXC6","canonical_record":{"source":{"id":"2401.16403","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T18:41:39Z","cross_cats_sorted":[],"title_canon_sha256":"d5c272e20a522041f3521a91a944a01161c21045612cd463a453c1f6e9123053","abstract_canon_sha256":"bb4f7aea421b57b669aced6dfcb4f635b3d9b42f8794ad1f4c95b8009d9c5b7e"},"schema_version":"1.0"},"canonical_sha256":"20ce63dc5e97e802614cd2698b2482534b972dd270814e0baea376cec91df17d","source":{"kind":"arxiv","id":"2401.16403","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.16403","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"arxiv_version","alias_value":"2401.16403v2","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.16403","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"pith_short_12","alias_value":"EDHGHXC6S7UA","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"pith_short_16","alias_value":"EDHGHXC6S7UAEYKM","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"pith_short_8","alias_value":"EDHGHXC6","created_at":"2026-07-05T07:39:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:EDHGHXC6S7UAEYKM2JUYWJECKN","target":"record","payload":{"canonical_record":{"source":{"id":"2401.16403","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T18:41:39Z","cross_cats_sorted":[],"title_canon_sha256":"d5c272e20a522041f3521a91a944a01161c21045612cd463a453c1f6e9123053","abstract_canon_sha256":"bb4f7aea421b57b669aced6dfcb4f635b3d9b42f8794ad1f4c95b8009d9c5b7e"},"schema_version":"1.0"},"canonical_sha256":"20ce63dc5e97e802614cd2698b2482534b972dd270814e0baea376cec91df17d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:39:34.777251Z","signature_b64":"5jEVdkroJSzYHXCMPovk+v0gxofjYunUAc3Kl/6RH4AhhWK6NicE4CDoHhRRQ/4rElor8RqMsfS3XqMpLWPQDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"20ce63dc5e97e802614cd2698b2482534b972dd270814e0baea376cec91df17d","last_reissued_at":"2026-07-05T07:39:34.776828Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:39:34.776828Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.16403","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:39:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xkdaXMRMl0yudJKs0f43Pf/01VgpxZeX+unJ00JzWgu+Wuexkdf4MladcoruW+7pP6LNfd6wXYW+rObOFMepBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T17:44:18.569941Z"},"content_sha256":"b1444a133849c4b7279a74505c1fa61162cb8bda1140a1a847db105038f9d919","schema_version":"1.0","event_id":"sha256:b1444a133849c4b7279a74505c1fa61162cb8bda1140a1a847db105038f9d919"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:EDHGHXC6S7UAEYKM2JUYWJECKN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"ViLexNorm: A Lexical Normalization Corpus for Vietnamese Social Media Text","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kiet Van Nguyen, Thanh-Nhi Nguyen, Thanh-Phong Le","submitted_at":"2024-01-29T18:41:39Z","abstract_excerpt":"Lexical normalization, a fundamental task in Natural Language Processing (NLP), involves the transformation of words into their canonical forms. This process has been proven to benefit various downstream NLP tasks greatly. In this work, we introduce Vietnamese Lexical Normalization (ViLexNorm), the first-ever corpus developed for the Vietnamese lexical normalization task. The corpus comprises over 10,000 pairs of sentences meticulously annotated by human annotators, sourced from public comments on Vietnam's most popular social media platforms. Various methods were used to evaluate our corpus, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.16403","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.16403/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:39:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GhrcyIrCvSfSBTib3uYTbHZp9Iv9ydsPBCBl4PgmWQ7J4VBAeyeZqpON7Hwg1R1bRgzZYliXy5AIQ9jnhHqQAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T17:44:18.570703Z"},"content_sha256":"701708851168ec9623ac9b97479071c950d43c7d4af81bc02e0c4b646b89c65b","schema_version":"1.0","event_id":"sha256:701708851168ec9623ac9b97479071c950d43c7d4af81bc02e0c4b646b89c65b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/EDHGHXC6S7UAEYKM2JUYWJECKN/bundle.json","state_url":"https://pith.science/pith/EDHGHXC6S7UAEYKM2JUYWJECKN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/EDHGHXC6S7UAEYKM2JUYWJECKN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-14T17:44:18Z","links":{"resolver":"https://pith.science/pith/EDHGHXC6S7UAEYKM2JUYWJECKN","bundle":"https://pith.science/pith/EDHGHXC6S7UAEYKM2JUYWJECKN/bundle.json","state":"https://pith.science/pith/EDHGHXC6S7UAEYKM2JUYWJECKN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/EDHGHXC6S7UAEYKM2JUYWJECKN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:EDHGHXC6S7UAEYKM2JUYWJECKN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"bb4f7aea421b57b669aced6dfcb4f635b3d9b42f8794ad1f4c95b8009d9c5b7e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T18:41:39Z","title_canon_sha256":"d5c272e20a522041f3521a91a944a01161c21045612cd463a453c1f6e9123053"},"schema_version":"1.0","source":{"id":"2401.16403","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.16403","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"arxiv_version","alias_value":"2401.16403v2","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.16403","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"pith_short_12","alias_value":"EDHGHXC6S7UA","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"pith_short_16","alias_value":"EDHGHXC6S7UAEYKM","created_at":"2026-07-05T07:39:34Z"},{"alias_kind":"pith_short_8","alias_value":"EDHGHXC6","created_at":"2026-07-05T07:39:34Z"}],"graph_snapshots":[{"event_id":"sha256:701708851168ec9623ac9b97479071c950d43c7d4af81bc02e0c4b646b89c65b","target":"graph","created_at":"2026-07-05T07:39:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.16403/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Lexical normalization, a fundamental task in Natural Language Processing (NLP), involves the transformation of words into their canonical forms. This process has been proven to benefit various downstream NLP tasks greatly. In this work, we introduce Vietnamese Lexical Normalization (ViLexNorm), the first-ever corpus developed for the Vietnamese lexical normalization task. The corpus comprises over 10,000 pairs of sentences meticulously annotated by human annotators, sourced from public comments on Vietnam's most popular social media platforms. Various methods were used to evaluate our corpus, ","authors_text":"Kiet Van Nguyen, Thanh-Nhi Nguyen, Thanh-Phong Le","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T18:41:39Z","title":"ViLexNorm: A Lexical Normalization Corpus for Vietnamese Social Media Text"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.16403","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b1444a133849c4b7279a74505c1fa61162cb8bda1140a1a847db105038f9d919","target":"record","created_at":"2026-07-05T07:39:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"bb4f7aea421b57b669aced6dfcb4f635b3d9b42f8794ad1f4c95b8009d9c5b7e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T18:41:39Z","title_canon_sha256":"d5c272e20a522041f3521a91a944a01161c21045612cd463a453c1f6e9123053"},"schema_version":"1.0","source":{"id":"2401.16403","kind":"arxiv","version":2}},"canonical_sha256":"20ce63dc5e97e802614cd2698b2482534b972dd270814e0baea376cec91df17d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"20ce63dc5e97e802614cd2698b2482534b972dd270814e0baea376cec91df17d","first_computed_at":"2026-07-05T07:39:34.776828Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:39:34.776828Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"5jEVdkroJSzYHXCMPovk+v0gxofjYunUAc3Kl/6RH4AhhWK6NicE4CDoHhRRQ/4rElor8RqMsfS3XqMpLWPQDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:39:34.777251Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.16403","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b1444a133849c4b7279a74505c1fa61162cb8bda1140a1a847db105038f9d919","sha256:701708851168ec9623ac9b97479071c950d43c7d4af81bc02e0c4b646b89c65b"],"state_sha256":"cf3671865819e22140c378be13ff207bceb805947c9a1190ae7daba84adb3de1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"m1zlZ+IId9/1hlT9w9+Gyzp14y3jPt2O2rFiqNK6jqDcKnjykgRVsiNv04Fz7jzzC2Lw5cRkFCRnuMjYgM+ABw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-14T17:44:18.576197Z","bundle_sha256":"1fd10d590a357a2bb3a473625dce50af93855ce1b206bcad8449029e969ca415"}}