{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FWHU4GRSIRWQEY3BRZFTGY4SDT","short_pith_number":"pith:FWHU4GRS","schema_version":"1.0","canonical_sha256":"2d8f4e1a32446d0263618e4b3363921cfe9d09a154a1579c8b44752c1418205b","source":{"kind":"arxiv","id":"2305.14483","version":1},"attestation_state":"computed","paper":{"title":"Language Model Self-improvement by Reinforcement Learning Contemplation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiacheng Xu, Jing-cheng Pang, Kaiyuan Li, Pengyuan Wang, Xiong-Hui Chen, Yang Yu, Zongzhang Zhang","submitted_at":"2023-05-23T19:25:52Z","abstract_excerpt":"Large Language Models (LLMs) have exhibited remarkable performance across various natural language processing (NLP) tasks. However, fine-tuning these models often necessitates substantial supervision, which can be expensive and time-consuming to obtain. This paper introduces a novel unsupervised method called LanguageModel Self-Improvement by Reinforcement Learning Contemplation (SIRLC) that improves LLMs without reliance on external labels. Our approach is grounded in the observation that it is simpler for language models to assess text quality than to generate text. Building on this insight,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14483","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T19:25:52Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2a1303631a00857a7874600015655f62142032dd4e71e65801a60ed4374c88a2","abstract_canon_sha256":"bfabae8865a7b066c2eacfbec0fc4a4515a46e7370d81773f634804865b158f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:13:24.054102Z","signature_b64":"3n6phAVzkG4Ug5yXLz66oybiaNVEWPYuX7veGIK0h6HOYzfK01TxBz6R2+xew89S2Xqu9+BoOk4M7DdOsZNFAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d8f4e1a32446d0263618e4b3363921cfe9d09a154a1579c8b44752c1418205b","last_reissued_at":"2026-07-05T06:13:24.053657Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:13:24.053657Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Model Self-improvement by Reinforcement Learning Contemplation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiacheng Xu, Jing-cheng Pang, Kaiyuan Li, Pengyuan Wang, Xiong-Hui Chen, Yang Yu, Zongzhang Zhang","submitted_at":"2023-05-23T19:25:52Z","abstract_excerpt":"Large Language Models (LLMs) have exhibited remarkable performance across various natural language processing (NLP) tasks. However, fine-tuning these models often necessitates substantial supervision, which can be expensive and time-consuming to obtain. This paper introduces a novel unsupervised method called LanguageModel Self-Improvement by Reinforcement Learning Contemplation (SIRLC) that improves LLMs without reliance on external labels. Our approach is grounded in the observation that it is simpler for language models to assess text quality than to generate text. Building on this insight,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14483","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14483","created_at":"2026-07-05T06:13:24.053713+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14483v1","created_at":"2026-07-05T06:13:24.053713+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14483","created_at":"2026-07-05T06:13:24.053713+00:00"},{"alias_kind":"pith_short_12","alias_value":"FWHU4GRSIRWQ","created_at":"2026-07-05T06:13:24.053713+00:00"},{"alias_kind":"pith_short_16","alias_value":"FWHU4GRSIRWQEY3B","created_at":"2026-07-05T06:13:24.053713+00:00"},{"alias_kind":"pith_short_8","alias_value":"FWHU4GRS","created_at":"2026-07-05T06:13:24.053713+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29625","citing_title":"Improving Collaborative Storytelling with a Multi-Agent Framework Based on Large Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20189","citing_title":"SOLAR: A Self-Optimizing Open-Ended Autonomous Agent for Lifelong Learning and Continual Adaptation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09395","citing_title":"Empowering VLMs for Few-Shot Multimodal Time Series Classification via Tailored Agentic Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07691","citing_title":"ORPO: Monolithic Preference Optimization without Reference Model","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09395","citing_title":"Empowering VLMs for Few-Shot Multimodal Time Series Classification via Tailored Agentic Reasoning","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT","json":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT.json","graph_json":"https://pith.science/api/pith-number/FWHU4GRSIRWQEY3BRZFTGY4SDT/graph.json","events_json":"https://pith.science/api/pith-number/FWHU4GRSIRWQEY3BRZFTGY4SDT/events.json","paper":"https://pith.science/paper/FWHU4GRS"},"agent_actions":{"view_html":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT","download_json":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT.json","view_paper":"https://pith.science/paper/FWHU4GRS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14483&json=true","fetch_graph":"https://pith.science/api/pith-number/FWHU4GRSIRWQEY3BRZFTGY4SDT/graph.json","fetch_events":"https://pith.science/api/pith-number/FWHU4GRSIRWQEY3BRZFTGY4SDT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT/action/storage_attestation","attest_author":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT/action/author_attestation","sign_citation":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT/action/citation_signature","submit_replication":"https://pith.science/pith/FWHU4GRSIRWQEY3BRZFTGY4SDT/action/replication_record"}},"created_at":"2026-07-05T06:13:24.053713+00:00","updated_at":"2026-07-05T06:13:24.053713+00:00"}