{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PV54SZXURCTDJU2DC7JQCYRRYG","short_pith_number":"pith:PV54SZXU","schema_version":"1.0","canonical_sha256":"7d7bc966f488a634d34317d3016231c1a4f73fce90dbdc63cbde9cf1be4b0c93","source":{"kind":"arxiv","id":"2309.03852","version":3},"attestation_state":"computed","paper":{"title":"FLM-101B: An Open LLM and How to Train It with $100K Budget","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Bowen Qin, Jing Li, Li Du, Peng Han, Siqi Fan, Xiang Li, Xin Jiang, Xuezhi Fang, Xuying Meng, Yequan Wang, Yiqun Yao, Zheng Zhang","submitted_at":"2023-09-07T17:07:36Z","abstract_excerpt":"Large language models (LLMs) are considered important approaches towards foundational machine intelligence, achieving remarkable success in Natural Language Processing and multimodal tasks, among others. However, the carbon footprints and financial costs originating from heavy pre-training computation is a non-negligible issue. Progressive training methods, inspired by the neurogenesis process that grows neural structures, have shown potential to accelerate LLM pre-training. However, the algorithms, implementation, and practices for progressively training LLMs beyond 100B parameters remain und"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.03852","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-07T17:07:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ccb1f2b927a5e182a381c50d0af9d7228a1d7dda2d92bd0853e93c77287d4042","abstract_canon_sha256":"44d04be5f99636af844306561042b1e542797315b5a4f93beb51dc2c8e087357"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:33.291662Z","signature_b64":"UFnkXvvHr46ZuX8BpuGKuNT3hnNwnoOEwsi62lefnjkSDPYfYaEjzHBoctaJ+cuV0g3n/veqN14l6HFHi3mUAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7d7bc966f488a634d34317d3016231c1a4f73fce90dbdc63cbde9cf1be4b0c93","last_reissued_at":"2026-07-05T10:00:33.291206Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:33.291206Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FLM-101B: An Open LLM and How to Train It with $100K Budget","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Bowen Qin, Jing Li, Li Du, Peng Han, Siqi Fan, Xiang Li, Xin Jiang, Xuezhi Fang, Xuying Meng, Yequan Wang, Yiqun Yao, Zheng Zhang","submitted_at":"2023-09-07T17:07:36Z","abstract_excerpt":"Large language models (LLMs) are considered important approaches towards foundational machine intelligence, achieving remarkable success in Natural Language Processing and multimodal tasks, among others. However, the carbon footprints and financial costs originating from heavy pre-training computation is a non-negligible issue. Progressive training methods, inspired by the neurogenesis process that grows neural structures, have shown potential to accelerate LLM pre-training. However, the algorithms, implementation, and practices for progressively training LLMs beyond 100B parameters remain und"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.03852","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.03852/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.03852","created_at":"2026-07-05T10:00:33.291267+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.03852v3","created_at":"2026-07-05T10:00:33.291267+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.03852","created_at":"2026-07-05T10:00:33.291267+00:00"},{"alias_kind":"pith_short_12","alias_value":"PV54SZXURCTD","created_at":"2026-07-05T10:00:33.291267+00:00"},{"alias_kind":"pith_short_16","alias_value":"PV54SZXURCTDJU2D","created_at":"2026-07-05T10:00:33.291267+00:00"},{"alias_kind":"pith_short_8","alias_value":"PV54SZXU","created_at":"2026-07-05T10:00:33.291267+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.08008","citing_title":"Beyond Sunk Costs: Boosting LLM Pre-training Efficiency via Orthogonal Growth of Mixture-of-Experts","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":104,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG","json":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG.json","graph_json":"https://pith.science/api/pith-number/PV54SZXURCTDJU2DC7JQCYRRYG/graph.json","events_json":"https://pith.science/api/pith-number/PV54SZXURCTDJU2DC7JQCYRRYG/events.json","paper":"https://pith.science/paper/PV54SZXU"},"agent_actions":{"view_html":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG","download_json":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG.json","view_paper":"https://pith.science/paper/PV54SZXU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.03852&json=true","fetch_graph":"https://pith.science/api/pith-number/PV54SZXURCTDJU2DC7JQCYRRYG/graph.json","fetch_events":"https://pith.science/api/pith-number/PV54SZXURCTDJU2DC7JQCYRRYG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG/action/storage_attestation","attest_author":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG/action/author_attestation","sign_citation":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG/action/citation_signature","submit_replication":"https://pith.science/pith/PV54SZXURCTDJU2DC7JQCYRRYG/action/replication_record"}},"created_at":"2026-07-05T10:00:33.291267+00:00","updated_at":"2026-07-05T10:00:33.291267+00:00"}