{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KLN3W7RIMKWOENCFUSNZ5VMCYS","short_pith_number":"pith:KLN3W7RI","schema_version":"1.0","canonical_sha256":"52dbbb7e2862ace23445a49b9ed582c4b4ffc6375d7e3b171cd580ee40d3e911","source":{"kind":"arxiv","id":"2308.07645","version":2},"attestation_state":"computed","paper":{"title":"Steering Language Generation: Harnessing Contrastive Expert Guidance and Negative Prompting for Coherent and Diverse Synthetic Data Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Charles O'Neill, Ioana Ciuca, Jack Miller, Thang Bui, Yuan-Sen Ting","submitted_at":"2023-08-15T08:49:14Z","abstract_excerpt":"Large Language Models (LLMs) hold immense potential to generate synthetic data of high quality and utility, which has numerous applications from downstream model training to practical data utilisation. However, contemporary models, despite their impressive capacities, consistently struggle to produce both coherent and diverse data. To address the coherency issue, we introduce contrastive expert guidance, where the difference between the logit distributions of fine-tuned and base language models is emphasised to ensure domain adherence. In order to ensure diversity, we utilise existing real and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.07645","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-15T08:49:14Z","cross_cats_sorted":[],"title_canon_sha256":"6cc8433c2704312390b814e31c158bc56606c00e48c56a64bbfe8cbc899106ac","abstract_canon_sha256":"2f56f72ce22a4edc47cebcc6029c284c3ffb9ba9686a2994ca48021f6031fa19"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:42:07.597957Z","signature_b64":"lFF+YeekqolamS3n3KrndqV2imuaFdXrFnFe7RnvMIKUelsx9T1jkILKYOfqr0qM8ml9vcLjzgslBIpauUy7Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52dbbb7e2862ace23445a49b9ed582c4b4ffc6375d7e3b171cd580ee40d3e911","last_reissued_at":"2026-07-05T06:42:07.597541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:42:07.597541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Steering Language Generation: Harnessing Contrastive Expert Guidance and Negative Prompting for Coherent and Diverse Synthetic Data Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Charles O'Neill, Ioana Ciuca, Jack Miller, Thang Bui, Yuan-Sen Ting","submitted_at":"2023-08-15T08:49:14Z","abstract_excerpt":"Large Language Models (LLMs) hold immense potential to generate synthetic data of high quality and utility, which has numerous applications from downstream model training to practical data utilisation. However, contemporary models, despite their impressive capacities, consistently struggle to produce both coherent and diverse data. To address the coherency issue, we introduce contrastive expert guidance, where the difference between the logit distributions of fine-tuned and base language models is emphasised to ensure domain adherence. In order to ensure diversity, we utilise existing real and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.07645","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.07645/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.07645","created_at":"2026-07-05T06:42:07.597597+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.07645v2","created_at":"2026-07-05T06:42:07.597597+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.07645","created_at":"2026-07-05T06:42:07.597597+00:00"},{"alias_kind":"pith_short_12","alias_value":"KLN3W7RIMKWO","created_at":"2026-07-05T06:42:07.597597+00:00"},{"alias_kind":"pith_short_16","alias_value":"KLN3W7RIMKWOENCF","created_at":"2026-07-05T06:42:07.597597+00:00"},{"alias_kind":"pith_short_8","alias_value":"KLN3W7RI","created_at":"2026-07-05T06:42:07.597597+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29198","citing_title":"Guidance Contrastive Token Credit Assignment for Discrete Policy Optimization","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS","json":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS.json","graph_json":"https://pith.science/api/pith-number/KLN3W7RIMKWOENCFUSNZ5VMCYS/graph.json","events_json":"https://pith.science/api/pith-number/KLN3W7RIMKWOENCFUSNZ5VMCYS/events.json","paper":"https://pith.science/paper/KLN3W7RI"},"agent_actions":{"view_html":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS","download_json":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS.json","view_paper":"https://pith.science/paper/KLN3W7RI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.07645&json=true","fetch_graph":"https://pith.science/api/pith-number/KLN3W7RIMKWOENCFUSNZ5VMCYS/graph.json","fetch_events":"https://pith.science/api/pith-number/KLN3W7RIMKWOENCFUSNZ5VMCYS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS/action/storage_attestation","attest_author":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS/action/author_attestation","sign_citation":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS/action/citation_signature","submit_replication":"https://pith.science/pith/KLN3W7RIMKWOENCFUSNZ5VMCYS/action/replication_record"}},"created_at":"2026-07-05T06:42:07.597597+00:00","updated_at":"2026-07-05T06:42:07.597597+00:00"}