{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:H4LD6AU3H54YKDX5YR5YIAFJXI","short_pith_number":"pith:H4LD6AU3","canonical_record":{"source":{"id":"2404.14219","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-22T14:32:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8774e901ff8de25d5e3601bfa65aeb00131b7331dc8d2298a2f6a179e8e33c68","abstract_canon_sha256":"93b85d0213e496eebe98edad71a323787ea2fc29cd42a714a09774887c05dddb"},"schema_version":"1.0"},"canonical_sha256":"3f163f029b3f79850efdc47b8400a9ba163303fa97eaec69618e37ad597c07a6","source":{"kind":"arxiv","id":"2404.14219","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.14219","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"arxiv_version","alias_value":"2404.14219v4","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14219","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"pith_short_12","alias_value":"H4LD6AU3H54Y","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"pith_short_16","alias_value":"H4LD6AU3H54YKDX5","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"pith_short_8","alias_value":"H4LD6AU3","created_at":"2026-07-05T09:01:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:H4LD6AU3H54YKDX5YR5YIAFJXI","target":"record","payload":{"canonical_record":{"source":{"id":"2404.14219","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-22T14:32:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8774e901ff8de25d5e3601bfa65aeb00131b7331dc8d2298a2f6a179e8e33c68","abstract_canon_sha256":"93b85d0213e496eebe98edad71a323787ea2fc29cd42a714a09774887c05dddb"},"schema_version":"1.0"},"canonical_sha256":"3f163f029b3f79850efdc47b8400a9ba163303fa97eaec69618e37ad597c07a6","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:33.468014Z","signature_b64":"8kZmdxKTQYBKkc22ImvHrfT0j0t/gFvIg4xAdA0z+pnzQAEk1Q2aVbtnJt75YK1LKMwuJtl0zu3DeMrt2jGQDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f163f029b3f79850efdc47b8400a9ba163303fa97eaec69618e37ad597c07a6","last_reissued_at":"2026-07-05T09:01:33.467491Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:33.467491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2404.14219","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:01:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9UdR5hHtdojY9LFhP35sB+IPGm/UtzvrWJVPUCrsuinezioQzeMPYFaCPSjDUph2BtNpXE22IsvP2s1u6IjgBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T13:01:17.441631Z"},"content_sha256":"9e7a9ba92b7eb6b5a0db93a0f40e08145afc10af6e56305774c6e54b2b19b02b","schema_version":"1.0","event_id":"sha256:9e7a9ba92b7eb6b5a0db93a0f40e08145afc10af6e56305774c6e54b2b19b02b"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:H4LD6AU3H54YKDX5YR5YIAFJXI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","license":"http://creativecommons.org/licenses/by/4.0/","headline":"A 3.8 billion parameter model matches the performance of much larger models like Mixtral 8x7B while running on a phone.","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abhishek Goswami, Adil Salim, Ahmed Awadallah, Ali Mahmoudzadeh, Allie Del Giorno, Alon Benhaim, Amin Saied, Amit Bahree, Amit Garg, Ammar Ahmad Awan, Andrea Tupini, Anh Nguyen, Arash Bakhtiari, Arindam Mitra, Barun Patra, Bin Xiao, Brandon Norick, Caio C\\'esar Teodoro Mendes, Can Xu, Ce Liu, Chen Liang, Chenruidong Zhang, Chong Luo, Chunyu Wang, Corby Rosset, Cyril Zhang, Daniel Perez-Becker, Dan Iter, David Majercak, Dong Chen, Dongdong Chen, Donghan Yu, Dongwoo Kim, Emman Haider, Fan Yang, Guanhua Wang, Gustavo de Rosa, Haiping Wu, Hany Awadalla, Hao Cheng, Hardik Modi, Harkirat Behl, Heyang Qin, Hiteshi Sharma, James R. Lee, Jamie Huynh, Jiahang Xu, Jianfeng Gao, Jianmin Bao, Jianwei Yang, Jianwen Zhang, Jilong Xue, Johan Bjorck, Junheng Hao, Jyoti Aneja, Lars Liden, Lev Kurilenko, Lijuan Wang, Liliang Ren, Li Lyna Zhang, Liyuan Liu, Lu Yuan, Mahoud Khademi, Marah Abdin, Marko Radmilac, Martin Cai, Masahiro Tanaka, Matthew Dixon, Matt Mazzola, Mei Gao, Mengchen Liu, Michael Santacroce, Michael Wyatt, Min Gao, Misha Bilenko, Mojan Javaheripi, Nguyen Bach, Nikos Karampatziakis, Ning Shang, Olatunji Ruwase, Olli Saarikivi, Parul Chopra, Philipp Witte, Piero Kauffmann, Piyush Madan, Praneetha Vaddamanu, Qin Cai, Rachel Ward, Reid Pryzant, Ronen Eldan, Russell J. Hewett, Sam Ade Jacobs, Sambudha Roy, S\\'ebastien Bubeck, Shital Shah, Shuohang Wang, Sonali Yadav, Suriya Gunasekar, Swadheen Shukla, Thomas Portet, Victor Fragoso, Vishrav Chaudhary, Weijian Xu, Weishung Liu, Weizhu Chen, Wen Wen, Wenxiang Hu, XiaoDong Liu, Xiaoxia Wu, Xia Song, Xihui Lin, Xin Jin, Xin Wang, Xiren Zhou, Xiyang Dai, Yelong Shen, Yen-Chun Chen, Yifan Yang, Yi-Ling Chen, Yin Tat Lee, Yi Zhang, Young Jin Kim, Yuanzhi Li, Yue Zhang, Yunan Zhang, Yunsheng Li, Yu Wang, Zeqi Lin, Ziyi Yang","submitted_at":"2024-04-22T14:32:33Z","abstract_excerpt":"We introduce phi-3-mini, a 3.8 billion parameter language model trained on 3.3 trillion tokens, whose overall performance, as measured by both academic benchmarks and internal testing, rivals that of models such as Mixtral 8x7B and GPT-3.5 (e.g., phi-3-mini achieves 69% on MMLU and 8.38 on MT-bench), despite being small enough to be deployed on a phone. Our training dataset is a scaled-up version of the one used for phi-2, composed of heavily filtered publicly available web data and synthetic data. The model is also further aligned for robustness, safety, and chat format. We also provide param"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"phi-3-mini achieves 69% on MMLU and 8.38 on MT-bench, rivaling Mixtral 8x7B and GPT-3.5 despite its small size; phi-3.5-MoE matches or exceeds similar-scale open models on reasoning, math, and code.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the heavily filtered web and synthetic training data produces genuine capability gains rather than benchmark-specific optimization or undetected contamination, as detailed data composition and decontamination steps are not specified in the abstract.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Phi-3-mini (3.8B params, 3.3T tokens) reaches 69% MMLU and 8.38 MT-bench, matching larger models, with scaled-up 7B/14B variants and phi-3.5 extensions for multilingual, MoE, and vision capabilities.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A 3.8 billion parameter model matches the performance of much larger models like Mixtral 8x7B while running on a phone.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"f41150caff47969f6a1663df8d395de185c12b2aaac774f40d43c577bdc36aed"},"source":{"id":"2404.14219","kind":"arxiv","version":4},"verdict":{"id":"da200aa5-6762-45cc-b9ff-b86a1b32a852","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T20:15:08.916274Z","strongest_claim":"phi-3-mini achieves 69% on MMLU and 8.38 on MT-bench, rivaling Mixtral 8x7B and GPT-3.5 despite its small size; phi-3.5-MoE matches or exceeds similar-scale open models on reasoning, math, and code.","one_line_summary":"Phi-3-mini (3.8B params, 3.3T tokens) reaches 69% MMLU and 8.38 MT-bench, matching larger models, with scaled-up 7B/14B variants and phi-3.5 extensions for multilingual, MoE, and vision capabilities.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the heavily filtered web and synthetic training data produces genuine capability gains rather than benchmark-specific optimization or undetected contamination, as detailed data composition and decontamination steps are not specified in the abstract.","pith_extraction_headline":"A 3.8 billion parameter model matches the performance of much larger models like Mixtral 8x7B while running on a phone."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14219/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":26,"sample":[{"doi":"","year":null,"title":"Program Synthesis with Large Language Models","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","ref_index":1,"cited_arxiv_id":"2108.07732","is_internal_anchor":true},{"doi":"","year":1911,"title":"PIQA: Reasoning about Physical Commonsense in Natural Language","work_id":"0d865a62-6376-4606-8d3a-eeb3b6e9ba6d","ref_index":2,"cited_arxiv_id":"1911.11641","is_internal_anchor":true},{"doi":"","year":null,"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","ref_index":3,"cited_arxiv_id":"2110.14168","is_internal_anchor":true},{"doi":"","year":2019,"title":"Boolq: Exploring the surprising difficulty of natural yes/no questions","work_id":"0be24034-99e0-4968-8660-335c9312df5f","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","work_id":"3714835e-c5a6-4d7e-950c-be44670ed9e6","ref_index":5,"cited_arxiv_id":"2404.16821","is_internal_anchor":true}],"resolved_work":26,"snapshot_sha256":"437761539dbdb31ed119c3abc0efd36a666854a2d4c0c94827afb5dc7e9cf3bb","internal_anchors":19},"formal_canon":{"evidence_count":2,"snapshot_sha256":"54dc1f2d2b332c307e3d38009bb7c232a152ad0f611690be70c55e34fff6b14b"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"da200aa5-6762-45cc-b9ff-b86a1b32a852"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:01:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"e3N/xXYM7lJ1cyB5L9id29fMQmVWtNL8tcJ+419NZq6htDVXPc/MXgJW2Fl5zYzdsiniEl/uXm0/oL8RorfEDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T13:01:17.442487Z"},"content_sha256":"47b97fa5a54a2fced85a6f7b4ca3c71e06a64fcf6b937befc6add7b9fee7099d","schema_version":"1.0","event_id":"sha256:47b97fa5a54a2fced85a6f7b4ca3c71e06a64fcf6b937befc6add7b9fee7099d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/H4LD6AU3H54YKDX5YR5YIAFJXI/bundle.json","state_url":"https://pith.science/pith/H4LD6AU3H54YKDX5YR5YIAFJXI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/H4LD6AU3H54YKDX5YR5YIAFJXI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T13:01:17Z","links":{"resolver":"https://pith.science/pith/H4LD6AU3H54YKDX5YR5YIAFJXI","bundle":"https://pith.science/pith/H4LD6AU3H54YKDX5YR5YIAFJXI/bundle.json","state":"https://pith.science/pith/H4LD6AU3H54YKDX5YR5YIAFJXI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/H4LD6AU3H54YKDX5YR5YIAFJXI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:H4LD6AU3H54YKDX5YR5YIAFJXI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"93b85d0213e496eebe98edad71a323787ea2fc29cd42a714a09774887c05dddb","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-22T14:32:33Z","title_canon_sha256":"8774e901ff8de25d5e3601bfa65aeb00131b7331dc8d2298a2f6a179e8e33c68"},"schema_version":"1.0","source":{"id":"2404.14219","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.14219","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"arxiv_version","alias_value":"2404.14219v4","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14219","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"pith_short_12","alias_value":"H4LD6AU3H54Y","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"pith_short_16","alias_value":"H4LD6AU3H54YKDX5","created_at":"2026-07-05T09:01:33Z"},{"alias_kind":"pith_short_8","alias_value":"H4LD6AU3","created_at":"2026-07-05T09:01:33Z"}],"graph_snapshots":[{"event_id":"sha256:47b97fa5a54a2fced85a6f7b4ca3c71e06a64fcf6b937befc6add7b9fee7099d","target":"graph","created_at":"2026-07-05T09:01:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"phi-3-mini achieves 69% on MMLU and 8.38 on MT-bench, rivaling Mixtral 8x7B and GPT-3.5 despite its small size; phi-3.5-MoE matches or exceeds similar-scale open models on reasoning, math, and code."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the heavily filtered web and synthetic training data produces genuine capability gains rather than benchmark-specific optimization or undetected contamination, as detailed data composition and decontamination steps are not specified in the abstract."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Phi-3-mini (3.8B params, 3.3T tokens) reaches 69% MMLU and 8.38 MT-bench, matching larger models, with scaled-up 7B/14B variants and phi-3.5 extensions for multilingual, MoE, and vision capabilities."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"A 3.8 billion parameter model matches the performance of much larger models like Mixtral 8x7B while running on a phone."}],"snapshot_sha256":"f41150caff47969f6a1663df8d395de185c12b2aaac774f40d43c577bdc36aed"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"54dc1f2d2b332c307e3d38009bb7c232a152ad0f611690be70c55e34fff6b14b"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2404.14219/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We introduce phi-3-mini, a 3.8 billion parameter language model trained on 3.3 trillion tokens, whose overall performance, as measured by both academic benchmarks and internal testing, rivals that of models such as Mixtral 8x7B and GPT-3.5 (e.g., phi-3-mini achieves 69% on MMLU and 8.38 on MT-bench), despite being small enough to be deployed on a phone. Our training dataset is a scaled-up version of the one used for phi-2, composed of heavily filtered publicly available web data and synthetic data. The model is also further aligned for robustness, safety, and chat format. We also provide param","authors_text":"Abhishek Goswami, Adil Salim, Ahmed Awadallah, Ali Mahmoudzadeh, Allie Del Giorno, Alon Benhaim, Amin Saied, Amit Bahree, Amit Garg, Ammar Ahmad Awan, Andrea Tupini, Anh Nguyen, Arash Bakhtiari, Arindam Mitra, Barun Patra, Bin Xiao, Brandon Norick, Caio C\\'esar Teodoro Mendes, Can Xu, Ce Liu, Chen Liang, Chenruidong Zhang, Chong Luo, Chunyu Wang, Corby Rosset, Cyril Zhang, Daniel Perez-Becker, Dan Iter, David Majercak, Dong Chen, Dongdong Chen, Donghan Yu, Dongwoo Kim, Emman Haider, Fan Yang, Guanhua Wang, Gustavo de Rosa, Haiping Wu, Hany Awadalla, Hao Cheng, Hardik Modi, Harkirat Behl, Heyang Qin, Hiteshi Sharma, James R. Lee, Jamie Huynh, Jiahang Xu, Jianfeng Gao, Jianmin Bao, Jianwei Yang, Jianwen Zhang, Jilong Xue, Johan Bjorck, Junheng Hao, Jyoti Aneja, Lars Liden, Lev Kurilenko, Lijuan Wang, Liliang Ren, Li Lyna Zhang, Liyuan Liu, Lu Yuan, Mahoud Khademi, Marah Abdin, Marko Radmilac, Martin Cai, Masahiro Tanaka, Matthew Dixon, Matt Mazzola, Mei Gao, Mengchen Liu, Michael Santacroce, Michael Wyatt, Min Gao, Misha Bilenko, Mojan Javaheripi, Nguyen Bach, Nikos Karampatziakis, Ning Shang, Olatunji Ruwase, Olli Saarikivi, Parul Chopra, Philipp Witte, Piero Kauffmann, Piyush Madan, Praneetha Vaddamanu, Qin Cai, Rachel Ward, Reid Pryzant, Ronen Eldan, Russell J. Hewett, Sam Ade Jacobs, Sambudha Roy, S\\'ebastien Bubeck, Shital Shah, Shuohang Wang, Sonali Yadav, Suriya Gunasekar, Swadheen Shukla, Thomas Portet, Victor Fragoso, Vishrav Chaudhary, Weijian Xu, Weishung Liu, Weizhu Chen, Wen Wen, Wenxiang Hu, XiaoDong Liu, Xiaoxia Wu, Xia Song, Xihui Lin, Xin Jin, Xin Wang, Xiren Zhou, Xiyang Dai, Yelong Shen, Yen-Chun Chen, Yifan Yang, Yi-Ling Chen, Yin Tat Lee, Yi Zhang, Young Jin Kim, Yuanzhi Li, Yue Zhang, Yunan Zhang, Yunsheng Li, Yu Wang, Zeqi Lin, Ziyi Yang","cross_cats":["cs.AI"],"headline":"A 3.8 billion parameter model matches the performance of much larger models like Mixtral 8x7B while running on a phone.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone"},"references":{"count":26,"internal_anchors":19,"resolved_work":26,"sample":[{"cited_arxiv_id":"2108.07732","doi":"","is_internal_anchor":true,"ref_index":1,"title":"Program Synthesis with Large Language Models","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","year":null},{"cited_arxiv_id":"1911.11641","doi":"","is_internal_anchor":true,"ref_index":2,"title":"PIQA: Reasoning about Physical Commonsense in Natural Language","work_id":"0d865a62-6376-4606-8d3a-eeb3b6e9ba6d","year":1911},{"cited_arxiv_id":"2110.14168","doi":"","is_internal_anchor":true,"ref_index":3,"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Boolq: Exploring the surprising difficulty of natural yes/no questions","work_id":"0be24034-99e0-4968-8660-335c9312df5f","year":2019},{"cited_arxiv_id":"2404.16821","doi":"","is_internal_anchor":true,"ref_index":5,"title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","work_id":"3714835e-c5a6-4d7e-950c-be44670ed9e6","year":null}],"snapshot_sha256":"437761539dbdb31ed119c3abc0efd36a666854a2d4c0c94827afb5dc7e9cf3bb"},"source":{"id":"2404.14219","kind":"arxiv","version":4},"verdict":{"created_at":"2026-05-10T20:15:08.916274Z","id":"da200aa5-6762-45cc-b9ff-b86a1b32a852","model_set":{"reader":"grok-4.3"},"one_line_summary":"Phi-3-mini (3.8B params, 3.3T tokens) reaches 69% MMLU and 8.38 MT-bench, matching larger models, with scaled-up 7B/14B variants and phi-3.5 extensions for multilingual, MoE, and vision capabilities.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"A 3.8 billion parameter model matches the performance of much larger models like Mixtral 8x7B while running on a phone.","strongest_claim":"phi-3-mini achieves 69% on MMLU and 8.38 on MT-bench, rivaling Mixtral 8x7B and GPT-3.5 despite its small size; phi-3.5-MoE matches or exceeds similar-scale open models on reasoning, math, and code.","weakest_assumption":"That the heavily filtered web and synthetic training data produces genuine capability gains rather than benchmark-specific optimization or undetected contamination, as detailed data composition and decontamination steps are not specified in the abstract."}},"verdict_id":"da200aa5-6762-45cc-b9ff-b86a1b32a852"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9e7a9ba92b7eb6b5a0db93a0f40e08145afc10af6e56305774c6e54b2b19b02b","target":"record","created_at":"2026-07-05T09:01:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"93b85d0213e496eebe98edad71a323787ea2fc29cd42a714a09774887c05dddb","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-22T14:32:33Z","title_canon_sha256":"8774e901ff8de25d5e3601bfa65aeb00131b7331dc8d2298a2f6a179e8e33c68"},"schema_version":"1.0","source":{"id":"2404.14219","kind":"arxiv","version":4}},"canonical_sha256":"3f163f029b3f79850efdc47b8400a9ba163303fa97eaec69618e37ad597c07a6","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3f163f029b3f79850efdc47b8400a9ba163303fa97eaec69618e37ad597c07a6","first_computed_at":"2026-07-05T09:01:33.467491Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:01:33.467491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"8kZmdxKTQYBKkc22ImvHrfT0j0t/gFvIg4xAdA0z+pnzQAEk1Q2aVbtnJt75YK1LKMwuJtl0zu3DeMrt2jGQDA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:01:33.468014Z","signed_message":"canonical_sha256_bytes"},"source_id":"2404.14219","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9e7a9ba92b7eb6b5a0db93a0f40e08145afc10af6e56305774c6e54b2b19b02b","sha256:47b97fa5a54a2fced85a6f7b4ca3c71e06a64fcf6b937befc6add7b9fee7099d"],"state_sha256":"7175f9aeebc4687239b5c413e08cedd024543dec8bccc0cc02749bda4304a044"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bEI9bGehJPH45v+ajtYMVeHzbCyz+jkI2IrL3tjnKYUKTQhCnutOIBmnu0GuTfacmVQllbc0jRfUpY6v7JAzDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T13:01:17.452333Z","bundle_sha256":"792c24283974820f884e467445d2573146425b964b2bcac15cc24c362aa01428"}}