{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:RW5HNGQ3VL4KSKFLC3YO332KQD","short_pith_number":"pith:RW5HNGQ3","canonical_record":{"source":{"id":"2505.23108","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T05:21:54Z","cross_cats_sorted":[],"title_canon_sha256":"f2fb15967338c36eaee30fc975e04b7d952066858780c318f8d9a7bf44940f82","abstract_canon_sha256":"ec847200e67eac13bfdcdc6299f2a5cb919a89b29a335d02c443d95c095e315a"},"schema_version":"1.0"},"canonical_sha256":"8dba769a1baaf8a928ab16f0edef4a80e31e562e9b6a591283f5d81b547dcbcc","source":{"kind":"arxiv","id":"2505.23108","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.23108","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"arxiv_version","alias_value":"2505.23108v1","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23108","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"pith_short_12","alias_value":"RW5HNGQ3VL4K","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"pith_short_16","alias_value":"RW5HNGQ3VL4KSKFL","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"pith_short_8","alias_value":"RW5HNGQ3","created_at":"2026-07-05T11:11:55Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:RW5HNGQ3VL4KSKFLC3YO332KQD","target":"record","payload":{"canonical_record":{"source":{"id":"2505.23108","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T05:21:54Z","cross_cats_sorted":[],"title_canon_sha256":"f2fb15967338c36eaee30fc975e04b7d952066858780c318f8d9a7bf44940f82","abstract_canon_sha256":"ec847200e67eac13bfdcdc6299f2a5cb919a89b29a335d02c443d95c095e315a"},"schema_version":"1.0"},"canonical_sha256":"8dba769a1baaf8a928ab16f0edef4a80e31e562e9b6a591283f5d81b547dcbcc","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:55.677510Z","signature_b64":"bUBw0UbhrxVVTKzrLxbWHN/k8pDO5iczbA4i5TB2KBbpkTKoi5QiEAI5vPX8aPcAcDHmiYOXazJGPDE0Sm25Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8dba769a1baaf8a928ab16f0edef4a80e31e562e9b6a591283f5d81b547dcbcc","last_reissued_at":"2026-07-05T11:11:55.677008Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:55.677008Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.23108","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:11:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iG9ZDNfye1iYiGJjtzbWGibxbBVw4pWbj93GsIBsuq1PCv28r91ZbYhr8Zq0m3FQsg0e/0XGKGI3Oafs3qXmBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T08:09:40.975296Z"},"content_sha256":"5051778f5ecf69502944d0b53d28f23f58d5aa08ef1859282638f42a1f6bd836","schema_version":"1.0","event_id":"sha256:5051778f5ecf69502944d0b53d28f23f58d5aa08ef1859282638f42a1f6bd836"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:RW5HNGQ3VL4KSKFLC3YO332KQD","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Generating Diverse Training Samples for Relation Extraction with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongliang Dai, Piji Li, Zexuan Li","submitted_at":"2025-05-29T05:21:54Z","abstract_excerpt":"Using Large Language Models (LLMs) to generate training data can potentially be a preferable way to improve zero or few-shot NLP tasks. However, many problems remain to be investigated for this direction. For the task of Relation Extraction (RE), we find that samples generated by directly prompting LLMs may easily have high structural similarities with each other. They tend to use a limited variety of phrasing while expressing the relation between a pair of entities. Therefore, in this paper, we study how to effectively improve the diversity of the training samples generated with LLMs for RE, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23108","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:11:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aPdk/a4v9I4lTuYTHLrfCSNfWnUDpVOSp4lp979LdvZ+FYjR9lQH63FqAboZxA62w7uDAdLsB397LalK1sAIBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T08:09:40.975833Z"},"content_sha256":"1bae4dcd99d2acfded0035c1a5ed074809610e0e651bd517698c4f7f6913ed2c","schema_version":"1.0","event_id":"sha256:1bae4dcd99d2acfded0035c1a5ed074809610e0e651bd517698c4f7f6913ed2c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RW5HNGQ3VL4KSKFLC3YO332KQD/bundle.json","state_url":"https://pith.science/pith/RW5HNGQ3VL4KSKFLC3YO332KQD/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RW5HNGQ3VL4KSKFLC3YO332KQD/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T08:09:40Z","links":{"resolver":"https://pith.science/pith/RW5HNGQ3VL4KSKFLC3YO332KQD","bundle":"https://pith.science/pith/RW5HNGQ3VL4KSKFLC3YO332KQD/bundle.json","state":"https://pith.science/pith/RW5HNGQ3VL4KSKFLC3YO332KQD/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RW5HNGQ3VL4KSKFLC3YO332KQD/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:RW5HNGQ3VL4KSKFLC3YO332KQD","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ec847200e67eac13bfdcdc6299f2a5cb919a89b29a335d02c443d95c095e315a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T05:21:54Z","title_canon_sha256":"f2fb15967338c36eaee30fc975e04b7d952066858780c318f8d9a7bf44940f82"},"schema_version":"1.0","source":{"id":"2505.23108","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.23108","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"arxiv_version","alias_value":"2505.23108v1","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23108","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"pith_short_12","alias_value":"RW5HNGQ3VL4K","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"pith_short_16","alias_value":"RW5HNGQ3VL4KSKFL","created_at":"2026-07-05T11:11:55Z"},{"alias_kind":"pith_short_8","alias_value":"RW5HNGQ3","created_at":"2026-07-05T11:11:55Z"}],"graph_snapshots":[{"event_id":"sha256:1bae4dcd99d2acfded0035c1a5ed074809610e0e651bd517698c4f7f6913ed2c","target":"graph","created_at":"2026-07-05T11:11:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.23108/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Using Large Language Models (LLMs) to generate training data can potentially be a preferable way to improve zero or few-shot NLP tasks. However, many problems remain to be investigated for this direction. For the task of Relation Extraction (RE), we find that samples generated by directly prompting LLMs may easily have high structural similarities with each other. They tend to use a limited variety of phrasing while expressing the relation between a pair of entities. Therefore, in this paper, we study how to effectively improve the diversity of the training samples generated with LLMs for RE, ","authors_text":"Hongliang Dai, Piji Li, Zexuan Li","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T05:21:54Z","title":"Generating Diverse Training Samples for Relation Extraction with Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23108","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5051778f5ecf69502944d0b53d28f23f58d5aa08ef1859282638f42a1f6bd836","target":"record","created_at":"2026-07-05T11:11:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ec847200e67eac13bfdcdc6299f2a5cb919a89b29a335d02c443d95c095e315a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T05:21:54Z","title_canon_sha256":"f2fb15967338c36eaee30fc975e04b7d952066858780c318f8d9a7bf44940f82"},"schema_version":"1.0","source":{"id":"2505.23108","kind":"arxiv","version":1}},"canonical_sha256":"8dba769a1baaf8a928ab16f0edef4a80e31e562e9b6a591283f5d81b547dcbcc","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8dba769a1baaf8a928ab16f0edef4a80e31e562e9b6a591283f5d81b547dcbcc","first_computed_at":"2026-07-05T11:11:55.677008Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:11:55.677008Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"bUBw0UbhrxVVTKzrLxbWHN/k8pDO5iczbA4i5TB2KBbpkTKoi5QiEAI5vPX8aPcAcDHmiYOXazJGPDE0Sm25Aw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:11:55.677510Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.23108","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5051778f5ecf69502944d0b53d28f23f58d5aa08ef1859282638f42a1f6bd836","sha256:1bae4dcd99d2acfded0035c1a5ed074809610e0e651bd517698c4f7f6913ed2c"],"state_sha256":"09d64f438b64c655c020698e64daf477879a65cfeaf4b2fbcca74eaaea6404b5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"S1qgraGP2pVWAhSGwR6qlJWDmb7Bot7UVP6+XiZj3hLgB8OjFbiBf03NxWnWZB9lBQTz0npkiaOFMOWNc6mWDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T08:09:40.980606Z","bundle_sha256":"219c18eefe89abe033d39479a113e3f514fea48c489c31c52687a2ecb0679405"}}