{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:APG26TV33VAS5KQS53VD6B73FB","short_pith_number":"pith:APG26TV3","schema_version":"1.0","canonical_sha256":"03cdaf4ebbdd412eaa12eeea3f07fb2876cc680cfd6b436a73f8efae43ff8d73","source":{"kind":"arxiv","id":"2503.00812","version":2},"attestation_state":"computed","paper":{"title":"BOSE: A Systematic Evaluation Method Optimized for Base Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Changxin Tian, Hongzhi Luan, Jun Zhou, Kunlong Chen, Xiaolu Zhang, Zhaoxin Huan, Zhiqiang Zhang","submitted_at":"2025-03-02T09:38:12Z","abstract_excerpt":"This paper poses two critical issues in evaluating base models (without post-training): (1) Unstable evaluation during training: in the early stages of pre-training, the models lack the capability to answer questions as required, leading to unstable evaluation results. This instability makes it difficult to provide solid conclusions to guide the training, especially for key experiments such as data ablation and scaling law. (2) Inconsistency between base and instruct models: base models generally exhibit poorer evaluation performance compared to corresponding instruct models. This gap poses a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00812","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-02T09:38:12Z","cross_cats_sorted":[],"title_canon_sha256":"ce898489c385a118ac9a0de7f7395d193db2d0a716c037321787c9b62b9db507","abstract_canon_sha256":"3fbabd9c6d4c7e83faf0f6b25457278190934d327cef1c9aa7d6a7e57209ab83"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:28.660548Z","signature_b64":"JnHtOpf03R+6bYSqwyBHH1tnkARys5hnK1KXBusuSGVBWuU1M8JxP9RFV6vkz+pH3PBfoC47ZzDwG2t72kuRCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03cdaf4ebbdd412eaa12eeea3f07fb2876cc680cfd6b436a73f8efae43ff8d73","last_reissued_at":"2026-07-05T11:12:28.660005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:28.660005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BOSE: A Systematic Evaluation Method Optimized for Base Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Changxin Tian, Hongzhi Luan, Jun Zhou, Kunlong Chen, Xiaolu Zhang, Zhaoxin Huan, Zhiqiang Zhang","submitted_at":"2025-03-02T09:38:12Z","abstract_excerpt":"This paper poses two critical issues in evaluating base models (without post-training): (1) Unstable evaluation during training: in the early stages of pre-training, the models lack the capability to answer questions as required, leading to unstable evaluation results. This instability makes it difficult to provide solid conclusions to guide the training, especially for key experiments such as data ablation and scaling law. (2) Inconsistency between base and instruct models: base models generally exhibit poorer evaluation performance compared to corresponding instruct models. This gap poses a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00812","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00812/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00812","created_at":"2026-07-05T11:12:28.660061+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00812v2","created_at":"2026-07-05T11:12:28.660061+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00812","created_at":"2026-07-05T11:12:28.660061+00:00"},{"alias_kind":"pith_short_12","alias_value":"APG26TV33VAS","created_at":"2026-07-05T11:12:28.660061+00:00"},{"alias_kind":"pith_short_16","alias_value":"APG26TV33VAS5KQS","created_at":"2026-07-05T11:12:28.660061+00:00"},{"alias_kind":"pith_short_8","alias_value":"APG26TV3","created_at":"2026-07-05T11:12:28.660061+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15079","citing_title":"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02650","citing_title":"Revealing the Learning Dynamics of Long-Context Continual Pre-training","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB","json":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB.json","graph_json":"https://pith.science/api/pith-number/APG26TV33VAS5KQS53VD6B73FB/graph.json","events_json":"https://pith.science/api/pith-number/APG26TV33VAS5KQS53VD6B73FB/events.json","paper":"https://pith.science/paper/APG26TV3"},"agent_actions":{"view_html":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB","download_json":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB.json","view_paper":"https://pith.science/paper/APG26TV3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00812&json=true","fetch_graph":"https://pith.science/api/pith-number/APG26TV33VAS5KQS53VD6B73FB/graph.json","fetch_events":"https://pith.science/api/pith-number/APG26TV33VAS5KQS53VD6B73FB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB/action/storage_attestation","attest_author":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB/action/author_attestation","sign_citation":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB/action/citation_signature","submit_replication":"https://pith.science/pith/APG26TV33VAS5KQS53VD6B73FB/action/replication_record"}},"created_at":"2026-07-05T11:12:28.660061+00:00","updated_at":"2026-07-05T11:12:28.660061+00:00"}