{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BQEWLZIIFRMSEI22BJGIO3YD7Y","short_pith_number":"pith:BQEWLZII","schema_version":"1.0","canonical_sha256":"0c0965e5082c5922235a0a4c876f03fe16e1bb0cea9c1823866d3d3bf4e3f5c9","source":{"kind":"arxiv","id":"2409.13831","version":1},"attestation_state":"computed","paper":{"title":"Measuring Copyright Risks of Large Language Model via Partial Information Probing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.CL","authors_text":"Denghui Zhang, Huajie Shao, Suzhen Duan, Weijie Zhao, Zhaozhuo Xu","submitted_at":"2024-09-20T18:16:05Z","abstract_excerpt":"Exploring the data sources used to train Large Language Models (LLMs) is a crucial direction in investigating potential copyright infringement by these models. While this approach can identify the possible use of copyrighted materials in training data, it does not directly measure infringing risks. Recent research has shifted towards testing whether LLMs can directly output copyrighted content. Addressing this direction, we investigate and assess LLMs' capacity to generate infringing content by providing them with partial information from copyrighted materials, and try to use iterative prompti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13831","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-20T18:16:05Z","cross_cats_sorted":["cs.AI","cs.CR"],"title_canon_sha256":"73f21f31e9553b8cc8e67809228dc42c8f3cf30e33bc499cae320d49287faf7c","abstract_canon_sha256":"89ddce74a4d19a2447b94deabd3fbd6243828d0000d52645b76d066fe71390b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:07.335348Z","signature_b64":"pvjHLe5jKafPzkvHjt/Nxn0hokWnD0znxeElocNVR2pAtUPcaBm/ggRqtlOjvx4wgto6+PUpP03unTmSYoRvBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c0965e5082c5922235a0a4c876f03fe16e1bb0cea9c1823866d3d3bf4e3f5c9","last_reissued_at":"2026-07-05T09:10:07.334809Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:07.334809Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring Copyright Risks of Large Language Model via Partial Information Probing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.CL","authors_text":"Denghui Zhang, Huajie Shao, Suzhen Duan, Weijie Zhao, Zhaozhuo Xu","submitted_at":"2024-09-20T18:16:05Z","abstract_excerpt":"Exploring the data sources used to train Large Language Models (LLMs) is a crucial direction in investigating potential copyright infringement by these models. While this approach can identify the possible use of copyrighted materials in training data, it does not directly measure infringing risks. Recent research has shifted towards testing whether LLMs can directly output copyrighted content. Addressing this direction, we investigate and assess LLMs' capacity to generate infringing content by providing them with partial information from copyrighted materials, and try to use iterative prompti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13831","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13831/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13831","created_at":"2026-07-05T09:10:07.334874+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13831v1","created_at":"2026-07-05T09:10:07.334874+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13831","created_at":"2026-07-05T09:10:07.334874+00:00"},{"alias_kind":"pith_short_12","alias_value":"BQEWLZIIFRMS","created_at":"2026-07-05T09:10:07.334874+00:00"},{"alias_kind":"pith_short_16","alias_value":"BQEWLZIIFRMSEI22","created_at":"2026-07-05T09:10:07.334874+00:00"},{"alias_kind":"pith_short_8","alias_value":"BQEWLZII","created_at":"2026-07-05T09:10:07.334874+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.17767","citing_title":"ISACL: Internal State Analyzer for Copyrighted Training Data Leakage","ref_index":70,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y","json":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y.json","graph_json":"https://pith.science/api/pith-number/BQEWLZIIFRMSEI22BJGIO3YD7Y/graph.json","events_json":"https://pith.science/api/pith-number/BQEWLZIIFRMSEI22BJGIO3YD7Y/events.json","paper":"https://pith.science/paper/BQEWLZII"},"agent_actions":{"view_html":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y","download_json":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y.json","view_paper":"https://pith.science/paper/BQEWLZII","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13831&json=true","fetch_graph":"https://pith.science/api/pith-number/BQEWLZIIFRMSEI22BJGIO3YD7Y/graph.json","fetch_events":"https://pith.science/api/pith-number/BQEWLZIIFRMSEI22BJGIO3YD7Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y/action/storage_attestation","attest_author":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y/action/author_attestation","sign_citation":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y/action/citation_signature","submit_replication":"https://pith.science/pith/BQEWLZIIFRMSEI22BJGIO3YD7Y/action/replication_record"}},"created_at":"2026-07-05T09:10:07.334874+00:00","updated_at":"2026-07-05T09:10:07.334874+00:00"}