{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZDQBXWEGQDYQY6437GCLY3WWOA","short_pith_number":"pith:ZDQBXWEG","schema_version":"1.0","canonical_sha256":"c8e01bd88680f10c7b9bf984bc6ed6702dfdace4a0a6cb268e5012843b29e4d9","source":{"kind":"arxiv","id":"2311.08721","version":2},"attestation_state":"computed","paper":{"title":"A Robust Semantics-based Watermark for Large Language Model against Paraphrasing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Dawei Yin, Han Xu, Jie Ren, Jiliang Tang, Shuaiqiang Wang, Yiding Liu, Yingqian Cui","submitted_at":"2023-11-15T06:19:02Z","abstract_excerpt":"Large language models (LLMs) have show great ability in various natural language tasks. However, there are concerns that LLMs are possible to be used improperly or even illegally. To prevent the malicious usage of LLMs, detecting LLM-generated text becomes crucial in the deployment of LLM applications. Watermarking is an effective strategy to detect the LLM-generated content by encoding a pre-defined secret watermark to facilitate the detection process. However, the majority of existing watermark methods leverage the simple hashes of precedent tokens to partition vocabulary. Such watermark can"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.08721","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2023-11-15T06:19:02Z","cross_cats_sorted":[],"title_canon_sha256":"155912cbbb8b323f2c64b413492f96da34f081d28db747c4bfa6a9abd252c547","abstract_canon_sha256":"691b35f0a8db516d441a41a87b5ab838bff9558632d767b5fa9be618cc758043"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:47.917831Z","signature_b64":"3EsLSA5rpbwA27AGHfir8GhV5nvXiG18lE88tVsLeSZmrx4qk0qs14OPPfuoBkWxMCOXCHS0REAlwNrdvl5tCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c8e01bd88680f10c7b9bf984bc6ed6702dfdace4a0a6cb268e5012843b29e4d9","last_reissued_at":"2026-07-05T08:02:47.917343Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:47.917343Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Robust Semantics-based Watermark for Large Language Model against Paraphrasing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Dawei Yin, Han Xu, Jie Ren, Jiliang Tang, Shuaiqiang Wang, Yiding Liu, Yingqian Cui","submitted_at":"2023-11-15T06:19:02Z","abstract_excerpt":"Large language models (LLMs) have show great ability in various natural language tasks. However, there are concerns that LLMs are possible to be used improperly or even illegally. To prevent the malicious usage of LLMs, detecting LLM-generated text becomes crucial in the deployment of LLM applications. Watermarking is an effective strategy to detect the LLM-generated content by encoding a pre-defined secret watermark to facilitate the detection process. However, the majority of existing watermark methods leverage the simple hashes of precedent tokens to partition vocabulary. Such watermark can"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.08721","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.08721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.08721","created_at":"2026-07-05T08:02:47.917395+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.08721v2","created_at":"2026-07-05T08:02:47.917395+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.08721","created_at":"2026-07-05T08:02:47.917395+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZDQBXWEGQDYQ","created_at":"2026-07-05T08:02:47.917395+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZDQBXWEGQDYQY643","created_at":"2026-07-05T08:02:47.917395+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZDQBXWEG","created_at":"2026-07-05T08:02:47.917395+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25796","citing_title":"SAMark: A Self-Anchored Text Watermarking with Paragraph-Level Paraphrase Robustness","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2508.11548","citing_title":"Copyright Protection for Large Language Models: A Survey of Methods, Challenges, and Trends","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18333","citing_title":"Position: LLM Watermarking Should Align Stakeholders' Incentives for Practical Adoption","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08964","citing_title":"Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05443","citing_title":"SLAM: Structural Linguistic Activation Marking for Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04305","citing_title":"SWAN: Semantic Watermarking with Abstract Meaning Representation","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA","json":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA.json","graph_json":"https://pith.science/api/pith-number/ZDQBXWEGQDYQY6437GCLY3WWOA/graph.json","events_json":"https://pith.science/api/pith-number/ZDQBXWEGQDYQY6437GCLY3WWOA/events.json","paper":"https://pith.science/paper/ZDQBXWEG"},"agent_actions":{"view_html":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA","download_json":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA.json","view_paper":"https://pith.science/paper/ZDQBXWEG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.08721&json=true","fetch_graph":"https://pith.science/api/pith-number/ZDQBXWEGQDYQY6437GCLY3WWOA/graph.json","fetch_events":"https://pith.science/api/pith-number/ZDQBXWEGQDYQY6437GCLY3WWOA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA/action/storage_attestation","attest_author":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA/action/author_attestation","sign_citation":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA/action/citation_signature","submit_replication":"https://pith.science/pith/ZDQBXWEGQDYQY6437GCLY3WWOA/action/replication_record"}},"created_at":"2026-07-05T08:02:47.917395+00:00","updated_at":"2026-07-05T08:02:47.917395+00:00"}