{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7NXWRYFS36RY6G63OK7WIYXOV7","short_pith_number":"pith:7NXWRYFS","schema_version":"1.0","canonical_sha256":"fb6f68e0b2dfa38f1bdb72bf6462eeafc83af6f8be5b14fa297050b8510aeb4a","source":{"kind":"arxiv","id":"2305.03047","version":2},"attestation_state":"computed","paper":{"title":"Principle-Driven Self-Alignment of Language Models from Scratch with Minimal Human Supervision","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Chuang Gan, David Cox, Hongxin Zhang, Qinhong Zhou, Yikang Shen, Yiming Yang, Zhenfang Chen, Zhiqing Sun","submitted_at":"2023-05-04T17:59:28Z","abstract_excerpt":"Recent AI-assistant agents, such as ChatGPT, predominantly rely on supervised fine-tuning (SFT) with human annotations and reinforcement learning from human feedback (RLHF) to align the output of large language models (LLMs) with human intentions, ensuring they are helpful, ethical, and reliable. However, this dependence can significantly constrain the true potential of AI-assistant agents due to the high cost of obtaining human supervision and the related issues on quality, reliability, diversity, self-consistency, and undesirable biases. To address these challenges, we propose a novel approa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.03047","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-05-04T17:59:28Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY"],"title_canon_sha256":"cde4f46f4f4418bbb0ea45ada2dd300417db3652a663eab8279855bbd5cdcac1","abstract_canon_sha256":"5fee1d0b5bcc2eed64d9d67122dca64ffaf592cd771df36758ed04a17e16df6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:19:19.839246Z","signature_b64":"a6Zp91PXWIdAG9NAKpiXcCN5I7ItQwTdXE4pfUPYB/j0I4Ya0PtFCzHUD6Fa8NloQpRkefTivgNGRO5hgwTVAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb6f68e0b2dfa38f1bdb72bf6462eeafc83af6f8be5b14fa297050b8510aeb4a","last_reissued_at":"2026-07-05T07:19:19.838735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:19:19.838735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Principle-Driven Self-Alignment of Language Models from Scratch with Minimal Human Supervision","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Chuang Gan, David Cox, Hongxin Zhang, Qinhong Zhou, Yikang Shen, Yiming Yang, Zhenfang Chen, Zhiqing Sun","submitted_at":"2023-05-04T17:59:28Z","abstract_excerpt":"Recent AI-assistant agents, such as ChatGPT, predominantly rely on supervised fine-tuning (SFT) with human annotations and reinforcement learning from human feedback (RLHF) to align the output of large language models (LLMs) with human intentions, ensuring they are helpful, ethical, and reliable. However, this dependence can significantly constrain the true potential of AI-assistant agents due to the high cost of obtaining human supervision and the related issues on quality, reliability, diversity, self-consistency, and undesirable biases. To address these challenges, we propose a novel approa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.03047","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.03047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.03047","created_at":"2026-07-05T07:19:19.838790+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.03047v2","created_at":"2026-07-05T07:19:19.838790+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.03047","created_at":"2026-07-05T07:19:19.838790+00:00"},{"alias_kind":"pith_short_12","alias_value":"7NXWRYFS36RY","created_at":"2026-07-05T07:19:19.838790+00:00"},{"alias_kind":"pith_short_16","alias_value":"7NXWRYFS36RY6G63","created_at":"2026-07-05T07:19:19.838790+00:00"},{"alias_kind":"pith_short_8","alias_value":"7NXWRYFS","created_at":"2026-07-05T07:19:19.838790+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2408.00724","citing_title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2305.17926","citing_title":"Large Language Models are not Fair Evaluators","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2309.14525","citing_title":"Aligning Large Multimodal Models with Factually Augmented RLHF","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2305.14233","citing_title":"Enhancing Chat Language Models by Scaling High-quality Instructional Conversations","ref_index":256,"is_internal_anchor":false},{"citing_arxiv_id":"2307.02483","citing_title":"Jailbroken: How Does LLM Safety Training Fail?","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2304.12244","citing_title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2310.03693","citing_title":"Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24544","citing_title":"STELLAR-E: a Synthetic, Tailored, End-to-end LLM Application Rigorous Evaluator","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2402.06196","citing_title":"Large Language Models: A Survey","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17651","citing_title":"Self-Refine: Iterative Refinement with Self-Feedback","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7","json":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7.json","graph_json":"https://pith.science/api/pith-number/7NXWRYFS36RY6G63OK7WIYXOV7/graph.json","events_json":"https://pith.science/api/pith-number/7NXWRYFS36RY6G63OK7WIYXOV7/events.json","paper":"https://pith.science/paper/7NXWRYFS"},"agent_actions":{"view_html":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7","download_json":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7.json","view_paper":"https://pith.science/paper/7NXWRYFS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.03047&json=true","fetch_graph":"https://pith.science/api/pith-number/7NXWRYFS36RY6G63OK7WIYXOV7/graph.json","fetch_events":"https://pith.science/api/pith-number/7NXWRYFS36RY6G63OK7WIYXOV7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7/action/storage_attestation","attest_author":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7/action/author_attestation","sign_citation":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7/action/citation_signature","submit_replication":"https://pith.science/pith/7NXWRYFS36RY6G63OK7WIYXOV7/action/replication_record"}},"created_at":"2026-07-05T07:19:19.838790+00:00","updated_at":"2026-07-05T07:19:19.838790+00:00"}