{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6TEATSL6QLM2AWREA5EAMOVQAR","short_pith_number":"pith:6TEATSL6","schema_version":"1.0","canonical_sha256":"f4c809c97e82d9a05a240748063ab004678feee2325d58e6e5b96de7905c03c7","source":{"kind":"arxiv","id":"2407.10627","version":1},"attestation_state":"computed","paper":{"title":"Arena Learning: Build Data Flywheel for LLMs Post-training via Simulated Chatbot Arena","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Can Xu, Haipeng Luo, Jianguang Lou, Pu Zhao, Qingfeng Sun, Qingwei Lin, Shifeng Chen, Weizhu Chen, Yansong Tang","submitted_at":"2024-07-15T11:26:07Z","abstract_excerpt":"Assessing the effectiveness of large language models (LLMs) presents substantial challenges. The method of conducting human-annotated battles in an online Chatbot Arena is a highly effective evaluative technique. However, this approach is limited by the costs and time required for human annotation. In this paper, we introduce Arena Learning, an innovative offline strategy designed to simulate these arena battles using AI-driven annotations to evaluate battle outcomes, thus facilitating the continuous improvement of the target model through both supervised fine-tuning and reinforcement learning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10627","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-15T11:26:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"0c9f3153895487114eb0b598fbda0ffcc24f76774315842508079afc3eea49a1","abstract_canon_sha256":"a2b4ebc6eee8c461194b7040303a533452ae0141325fcf99917b9f4f24a0ee67"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:02.168907Z","signature_b64":"pq4CXwTYiDgpwlD3HZNZ/fnEJX0FKPPVlyNE6fq02QxBR9sctOmy7hekFepyEOsNr2Z4oNhRtb6OnBopL5UbBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4c809c97e82d9a05a240748063ab004678feee2325d58e6e5b96de7905c03c7","last_reissued_at":"2026-07-05T08:44:02.168500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:02.168500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Arena Learning: Build Data Flywheel for LLMs Post-training via Simulated Chatbot Arena","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Can Xu, Haipeng Luo, Jianguang Lou, Pu Zhao, Qingfeng Sun, Qingwei Lin, Shifeng Chen, Weizhu Chen, Yansong Tang","submitted_at":"2024-07-15T11:26:07Z","abstract_excerpt":"Assessing the effectiveness of large language models (LLMs) presents substantial challenges. The method of conducting human-annotated battles in an online Chatbot Arena is a highly effective evaluative technique. However, this approach is limited by the costs and time required for human annotation. In this paper, we introduce Arena Learning, an innovative offline strategy designed to simulate these arena battles using AI-driven annotations to evaluate battle outcomes, thus facilitating the continuous improvement of the target model through both supervised fine-tuning and reinforcement learning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10627","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10627/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10627","created_at":"2026-07-05T08:44:02.168558+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10627v1","created_at":"2026-07-05T08:44:02.168558+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10627","created_at":"2026-07-05T08:44:02.168558+00:00"},{"alias_kind":"pith_short_12","alias_value":"6TEATSL6QLM2","created_at":"2026-07-05T08:44:02.168558+00:00"},{"alias_kind":"pith_short_16","alias_value":"6TEATSL6QLM2AWRE","created_at":"2026-07-05T08:44:02.168558+00:00"},{"alias_kind":"pith_short_8","alias_value":"6TEATSL6","created_at":"2026-07-05T08:44:02.168558+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19236","citing_title":"STARE: Surprisal-Guided Token-Level Advantage Reweighting for Policy Entropy Stability","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06462","citing_title":"Benchmark Everything Everywhere All at Once","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06230","citing_title":"Safactory: A Scalable Agentic Infrastructure for Training Trustworthy Autonomous Intelligence","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06230","citing_title":"Safactory: A Scalable Agentic Infrastructure for Training Trustworthy Autonomous Intelligence","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR","json":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR.json","graph_json":"https://pith.science/api/pith-number/6TEATSL6QLM2AWREA5EAMOVQAR/graph.json","events_json":"https://pith.science/api/pith-number/6TEATSL6QLM2AWREA5EAMOVQAR/events.json","paper":"https://pith.science/paper/6TEATSL6"},"agent_actions":{"view_html":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR","download_json":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR.json","view_paper":"https://pith.science/paper/6TEATSL6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10627&json=true","fetch_graph":"https://pith.science/api/pith-number/6TEATSL6QLM2AWREA5EAMOVQAR/graph.json","fetch_events":"https://pith.science/api/pith-number/6TEATSL6QLM2AWREA5EAMOVQAR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR/action/storage_attestation","attest_author":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR/action/author_attestation","sign_citation":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR/action/citation_signature","submit_replication":"https://pith.science/pith/6TEATSL6QLM2AWREA5EAMOVQAR/action/replication_record"}},"created_at":"2026-07-05T08:44:02.168558+00:00","updated_at":"2026-07-05T08:44:02.168558+00:00"}