{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YMMANWQN3DZDWSIO3SPEAHJR7T","short_pith_number":"pith:YMMANWQN","schema_version":"1.0","canonical_sha256":"c31806da0dd8f23b490edc9e401d31fcda79d3976edc33dabbbe9af77be64154","source":{"kind":"arxiv","id":"2404.16645","version":1},"attestation_state":"computed","paper":{"title":"Tele-FLM Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Bo Zhao, Chao Wang, Shuangyong Song, Tiejun Huang, Xiang Li, Xin Jiang, Xin Wang, Xinzhang Liu, Xuelong Li, Xuezhi Fang, Yequan Wang, Yiqun Yao, Yongxiang Li, Yuyao Huang, Yu Zhao, Zheng Zhang, Zhongjiang He, Zhongyuan Wang, Zihan Wang","submitted_at":"2024-04-25T14:34:47Z","abstract_excerpt":"Large language models (LLMs) have showcased profound capabilities in language understanding and generation, facilitating a wide array of applications. However, there is a notable paucity of detailed, open-sourced methodologies on efficiently scaling LLMs beyond 50 billion parameters with minimum trial-and-error cost and computational resources. In this report, we introduce Tele-FLM (aka FLM-2), a 52B open-sourced multilingual large language model that features a stable, efficient pre-training paradigm and enhanced factual judgment capabilities. Tele-FLM demonstrates superior multilingual langu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.16645","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-25T14:34:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"dcb9f1ad69bdb3cdf153da5d0e247f90ad5530a1a5e6547ed934f2c6e7f2a73d","abstract_canon_sha256":"f3705301c0c971fe4d06a0a9c8e868450590fe47d9bd6604813f667bee130121"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:12:10.878861Z","signature_b64":"hJvEzIpvSEDtSEQ/a2Xoo/eWadYt6JNmbQxlaT9p7aF4P4ThcETbm+AIf1D+1GhA10ZhYy+Gis+xshf2Lt9/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c31806da0dd8f23b490edc9e401d31fcda79d3976edc33dabbbe9af77be64154","last_reissued_at":"2026-07-05T08:12:10.878328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:12:10.878328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tele-FLM Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Bo Zhao, Chao Wang, Shuangyong Song, Tiejun Huang, Xiang Li, Xin Jiang, Xin Wang, Xinzhang Liu, Xuelong Li, Xuezhi Fang, Yequan Wang, Yiqun Yao, Yongxiang Li, Yuyao Huang, Yu Zhao, Zheng Zhang, Zhongjiang He, Zhongyuan Wang, Zihan Wang","submitted_at":"2024-04-25T14:34:47Z","abstract_excerpt":"Large language models (LLMs) have showcased profound capabilities in language understanding and generation, facilitating a wide array of applications. However, there is a notable paucity of detailed, open-sourced methodologies on efficiently scaling LLMs beyond 50 billion parameters with minimum trial-and-error cost and computational resources. In this report, we introduce Tele-FLM (aka FLM-2), a 52B open-sourced multilingual large language model that features a stable, efficient pre-training paradigm and enhanced factual judgment capabilities. Tele-FLM demonstrates superior multilingual langu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.16645","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.16645/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.16645","created_at":"2026-07-05T08:12:10.878402+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.16645v1","created_at":"2026-07-05T08:12:10.878402+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.16645","created_at":"2026-07-05T08:12:10.878402+00:00"},{"alias_kind":"pith_short_12","alias_value":"YMMANWQN3DZD","created_at":"2026-07-05T08:12:10.878402+00:00"},{"alias_kind":"pith_short_16","alias_value":"YMMANWQN3DZDWSIO","created_at":"2026-07-05T08:12:10.878402+00:00"},{"alias_kind":"pith_short_8","alias_value":"YMMANWQN","created_at":"2026-07-05T08:12:10.878402+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01392","citing_title":"Multi-Objective Exploration and Preference Optimization via Mutual Information","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03022","citing_title":"Hallucinations as Orthogonal Noise: Inference-Time Manifold Alignment via Dynamic Contextual Orthogonalization","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01923","citing_title":"Resonant Context Anchoring: Decoupling Attention Routing and Signal Gain at Inference Time","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T","json":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T.json","graph_json":"https://pith.science/api/pith-number/YMMANWQN3DZDWSIO3SPEAHJR7T/graph.json","events_json":"https://pith.science/api/pith-number/YMMANWQN3DZDWSIO3SPEAHJR7T/events.json","paper":"https://pith.science/paper/YMMANWQN"},"agent_actions":{"view_html":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T","download_json":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T.json","view_paper":"https://pith.science/paper/YMMANWQN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.16645&json=true","fetch_graph":"https://pith.science/api/pith-number/YMMANWQN3DZDWSIO3SPEAHJR7T/graph.json","fetch_events":"https://pith.science/api/pith-number/YMMANWQN3DZDWSIO3SPEAHJR7T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T/action/storage_attestation","attest_author":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T/action/author_attestation","sign_citation":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T/action/citation_signature","submit_replication":"https://pith.science/pith/YMMANWQN3DZDWSIO3SPEAHJR7T/action/replication_record"}},"created_at":"2026-07-05T08:12:10.878402+00:00","updated_at":"2026-07-05T08:12:10.878402+00:00"}