{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3YGJXA53ZXSOZKFFT5GFMQWTQE","short_pith_number":"pith:3YGJXA53","schema_version":"1.0","canonical_sha256":"de0c9b83bbcde4eca8a59f4c5642d381159ef1ca7fcf68afb1174166eddccb74","source":{"kind":"arxiv","id":"2307.16039","version":2},"attestation_state":"computed","paper":{"title":"Okapi: Instruction-tuned Large Language Models in Multiple Languages with Reinforcement Learning from Human Feedback","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chien Van Nguyen, Franck Dernoncourt, Nghia Trung Ngo, Ryan A. Rossi, Thien Huu Nguyen, Thuat Nguyen, Viet Dac Lai","submitted_at":"2023-07-29T18:01:46Z","abstract_excerpt":"A key technology for the development of large language models (LLMs) involves instruction tuning that helps align the models' responses with human expectations to realize impressive learning abilities. Two major approaches for instruction tuning characterize supervised fine-tuning (SFT) and reinforcement learning from human feedback (RLHF), which are currently applied to produce the best commercial LLMs (e.g., ChatGPT). To improve the accessibility of LLMs for research and development efforts, various instruction-tuned open-source LLMs have also been introduced recently, e.g., Alpaca, Vicuna, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.16039","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-07-29T18:01:46Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cfbb2baecbdf581a1e19a954ee75cd0d1a3d7437a9cf02623a220b19b1c4684b","abstract_canon_sha256":"5246c74a91ecd928bdbd6c14767af71671428faf88d8fd2a87536cc9f514561a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:36:55.649906Z","signature_b64":"kNMSgCIBxrHJ444hN+mKmVJvu4e8WVqo8wAmbdptmMkmOLznrAJiYz+wLAjr80Xenzf4PCIVbb2EoZCgpLk7CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de0c9b83bbcde4eca8a59f4c5642d381159ef1ca7fcf68afb1174166eddccb74","last_reissued_at":"2026-07-05T06:36:55.649421Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:36:55.649421Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Okapi: Instruction-tuned Large Language Models in Multiple Languages with Reinforcement Learning from Human Feedback","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chien Van Nguyen, Franck Dernoncourt, Nghia Trung Ngo, Ryan A. Rossi, Thien Huu Nguyen, Thuat Nguyen, Viet Dac Lai","submitted_at":"2023-07-29T18:01:46Z","abstract_excerpt":"A key technology for the development of large language models (LLMs) involves instruction tuning that helps align the models' responses with human expectations to realize impressive learning abilities. Two major approaches for instruction tuning characterize supervised fine-tuning (SFT) and reinforcement learning from human feedback (RLHF), which are currently applied to produce the best commercial LLMs (e.g., ChatGPT). To improve the accessibility of LLMs for research and development efforts, various instruction-tuned open-source LLMs have also been introduced recently, e.g., Alpaca, Vicuna, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.16039","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.16039/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.16039","created_at":"2026-07-05T06:36:55.649493+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.16039v2","created_at":"2026-07-05T06:36:55.649493+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.16039","created_at":"2026-07-05T06:36:55.649493+00:00"},{"alias_kind":"pith_short_12","alias_value":"3YGJXA53ZXSO","created_at":"2026-07-05T06:36:55.649493+00:00"},{"alias_kind":"pith_short_16","alias_value":"3YGJXA53ZXSOZKFF","created_at":"2026-07-05T06:36:55.649493+00:00"},{"alias_kind":"pith_short_8","alias_value":"3YGJXA53","created_at":"2026-07-05T06:36:55.649493+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.16154","citing_title":"AdaMCoT: Rethinking Cross-Lingual Factual Reasoning through Adaptive Multilingual Chain-of-Thought","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12917","citing_title":"Training Language Models to Self-Correct via Reinforcement Learning","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2309.00267","citing_title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","ref_index":80,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE","json":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE.json","graph_json":"https://pith.science/api/pith-number/3YGJXA53ZXSOZKFFT5GFMQWTQE/graph.json","events_json":"https://pith.science/api/pith-number/3YGJXA53ZXSOZKFFT5GFMQWTQE/events.json","paper":"https://pith.science/paper/3YGJXA53"},"agent_actions":{"view_html":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE","download_json":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE.json","view_paper":"https://pith.science/paper/3YGJXA53","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.16039&json=true","fetch_graph":"https://pith.science/api/pith-number/3YGJXA53ZXSOZKFFT5GFMQWTQE/graph.json","fetch_events":"https://pith.science/api/pith-number/3YGJXA53ZXSOZKFFT5GFMQWTQE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE/action/storage_attestation","attest_author":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE/action/author_attestation","sign_citation":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE/action/citation_signature","submit_replication":"https://pith.science/pith/3YGJXA53ZXSOZKFFT5GFMQWTQE/action/replication_record"}},"created_at":"2026-07-05T06:36:55.649493+00:00","updated_at":"2026-07-05T06:36:55.649493+00:00"}