{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:2LR4LPBOWIE25TP3CYFS33XNSD","short_pith_number":"pith:2LR4LPBO","canonical_record":{"source":{"id":"2408.00118","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-31T19:13:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f0c83b1d378b9b8629c78b542bb35f200a1e613dfaa0029096f206b3af1fd760","abstract_canon_sha256":"988d94bd4b98ba8eb221303a3c8a478ebeb329bc038811a13566339403e8f79b"},"schema_version":"1.0"},"canonical_sha256":"d2e3c5bc2eb209aecdfb160b2deeed90ff69a48a5f7c698d2e51441375eec723","source":{"kind":"arxiv","id":"2408.00118","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2408.00118","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"arxiv_version","alias_value":"2408.00118v3","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.00118","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"pith_short_12","alias_value":"2LR4LPBOWIE2","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"pith_short_16","alias_value":"2LR4LPBOWIE25TP3","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"pith_short_8","alias_value":"2LR4LPBO","created_at":"2026-07-05T09:14:36Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:2LR4LPBOWIE25TP3CYFS33XNSD","target":"record","payload":{"canonical_record":{"source":{"id":"2408.00118","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-31T19:13:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f0c83b1d378b9b8629c78b542bb35f200a1e613dfaa0029096f206b3af1fd760","abstract_canon_sha256":"988d94bd4b98ba8eb221303a3c8a478ebeb329bc038811a13566339403e8f79b"},"schema_version":"1.0"},"canonical_sha256":"d2e3c5bc2eb209aecdfb160b2deeed90ff69a48a5f7c698d2e51441375eec723","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:36.990964Z","signature_b64":"oTKuLa6oZ+kf1Ytg/UrYeeQzw1Wyr5kP8S+T3QXg9FkR24SK9a1qCBih31Hhh+9md5+juDtSmOP3mrPyJxnyCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d2e3c5bc2eb209aecdfb160b2deeed90ff69a48a5f7c698d2e51441375eec723","last_reissued_at":"2026-07-05T09:14:36.990462Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:36.990462Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2408.00118","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:14:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Qoip+DIcV8H2WhbBtEJ5UxrwNy97bgrZ2j0szitaAARLaHEzZ9KYugx51EIgKEvFHdhgozQ7f3HNNxPCpfiUCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T00:23:06.278988Z"},"content_sha256":"007f9290634d32cff2edb01ca4fdae81219f9148ba88caf4cc587bde798b5107","schema_version":"1.0","event_id":"sha256:007f9290634d32cff2edb01ca4fdae81219f9148ba88caf4cc587bde798b5107"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:2LR4LPBOWIE25TP3CYFS33XNSD","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Gemma 2: Improving Open Language Models at a Practical Size","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Gemma 2 models achieve leading performance at their sizes through interleaving local-global attention, group-query attention, and knowledge distillation on the smaller variants.","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abe Friesen, Alanna Walton, Alek Andreev, Alexandre Ram\\'e, Aliaksei Severyn, Alicia Parrish, Aliya Ahmad, Allen Hutchison, Alvin Abdagic, Amanda Carl, Amy Shen, Anand Rao, Anca Dragan, Andy Brock, Andy Coenen, Anthony Laforge, Antonia Paterson, Anton Tsitsulin, Armand Joulin, Behnam Neyshabur, Ben Bastian, Bilal Piot, Bobak Shahriari, Bo Wu, Brandon Royal, Cassidy Hardin, Charlie Chen, Charline Le Lan, Chintu Kumar, Chris Perry, Christopher A. Choquette-Choo, Chris Welty, Clement Farabet, Danila Sinopalnikov, David Weinberger, Demis Hassabis, Dimple Vijaykumar, Dominika Rogozi\\'nska, D. Sculley, Dustin Herbison, Elena Buchatskaya, Eli Collins, Elisa Bandy, Emma Wang, Erica Moreira, Eric Noland, Evan Senter, Evgenii Eltyshev, Francesco Visin, Gabriel Rasskin, Gary Wei, Gemma Team: Morgane Riviere, Glenn Cameron, Gus Martins, Hadi Hashemi, Hanna Klimczak-Pluci\\'nska, Harleen Batra, Harsh Dhand, Ivan Nardini, Jacinda Mein, Jack Zhou, James Svensson, Jean-bastien Grill, Jeanine Banks, Jeff Dean, Jeff Stanway, Jetha Chan, Jin Peng Zhou, Joana Carrasqueira, Joana Iljazi, Jocelyn Becker, Joe Fernandez, Joelle Barral, Johan Ferret, Joost van Amersfoort, Josh Gordon, Josh Lipschultz, Josh Newlan, Ju-yeong Ji, Kareem Mohamed, Kartikeya Badola, Kat Black, Kathleen Kenealy, Katie Millican, Keelin McDonell, Kelvin Nguyen, Kiranbir Sodhia, Kish Greene, Koray Kavukcuoglu, Lars Lowe Sjoesund, Laurent Sifre, Lauren Usui, Lena Heuermann, L\\'eonard Hussenot, Leticia Lago, Lilly McNealus, Livio Baldini Soares, Logan Kilpatrick, Lucas Dixon, Luciano Martins, Ludovic Peran, Machel Reid, Manvinder Singh, Mark Iverson, Martin G\\\"orner, Mateo Wirth, Matt Davidow, Matthew Rahtz, Matthew Watson, Matt Hoffman, Matt Miller, Mat Velloso, Meg Risdal, Mehran Kazemi, Michael Moynihan, Michelle Casbon, Ming Zhang, Minh Giang, Minsuk Kahng, Minwoo Park, Mofi Rahman, Mohit Khatwani, Natalie Dao, Nenshad Bardoliwalla, Nesh Devanathan, Neta Dumai, Nikola Momchev, Nilay Chauhan, Nino Vieillard, Noah Fiedel, Olivier Bachem, Oriol Vinyals, Oscar Wahltinez, Pankil Botarda, Parker Barnes, Paul Barham, Paul Michel, Pengchong Jin, Peter Liu, Petko Georgiev, Phil Culliton, Phoebe Kirk, Pier Giuseppe Sessa, Piotr Stanczyk, Pouya Tafti, Pradeep Kuppala, Raia Hadsell, Ramona Comanescu, Ramona Merhej, Ravin Kumar, Reena Jana, Reza Ardeshir Rokni, Rishabh Agarwal, Robert Dadashi, Ryan Mullins, Sabela Ramos, Samaneh Saadat, Sammy Jerome, Sarah Cogan, Sarah Perrin, Sara Mc Carthy, Sebastian Borgeaud, Sebastian Krause, S\\'ebastien M. R. Arnold, Sertan Girgin, Shantanu Thakoor, Shengyang Dai, Shreya Pathak, Shruti Garg, Shruti Sheth, Slav Petrov, Sue Ronstrom, Surya Bhupatiraju, Susan Chan, Thomas Mesnard, Timothy Jordan, Ting Yu, Tomas Kocisky, Tom Eccles, Tom Hennigan, Tris Warkentin, Tulsee Doshi, Victor Cotruta, Vihan Jain, Vikas Yadav, Vilobh Meshram, Vishal Dharmadhikari, Warren Barkley, Wei Wei, Wenming Ye, Woohyun Han, Woosuk Kwon, Xiang Xu, Zhe Shen, Zhitao Gong, Zichuan Wei, Zoubin Ghahramani","submitted_at":"2024-07-31T19:13:07Z","abstract_excerpt":"In this work, we introduce Gemma 2, a new addition to the Gemma family of lightweight, state-of-the-art open models, ranging in scale from 2 billion to 27 billion parameters. In this new version, we apply several known technical modifications to the Transformer architecture, such as interleaving local-global attentions (Beltagy et al., 2020a) and group-query attention (Ainslie et al., 2023). We also train the 2B and 9B models with knowledge distillation (Hinton et al., 2015) instead of next token prediction. The resulting models deliver the best performance for their size, and even offer compe"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"The resulting models deliver the best performance for their size, and even offer competitive alternatives to models that are 2-3 times bigger.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the reported performance gains are attributable to the listed architectural changes and distillation rather than to undisclosed differences in training data, compute, or evaluation protocols.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Gemma 2 models achieve leading performance at their sizes by combining established Transformer modifications with knowledge distillation for the 2B and 9B variants.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Gemma 2 models achieve leading performance at their sizes through interleaving local-global attention, group-query attention, and knowledge distillation on the smaller variants.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"f1176a06c6a324417e3b4d343bd1fc732f2f2c3a07552d7c91fb9f2ed3c972a7"},"source":{"id":"2408.00118","kind":"arxiv","version":3},"verdict":{"id":"305abfa3-83f3-4c36-8445-cb015f283e1f","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T12:06:11.309894Z","strongest_claim":"The resulting models deliver the best performance for their size, and even offer competitive alternatives to models that are 2-3 times bigger.","one_line_summary":"Gemma 2 models achieve leading performance at their sizes by combining established Transformer modifications with knowledge distillation for the 2B and 9B variants.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the reported performance gains are attributable to the listed architectural changes and distillation rather than to undisclosed differences in training data, compute, or evaluation protocols.","pith_extraction_headline":"Gemma 2 models achieve leading performance at their sizes through interleaving local-global attention, group-query attention, and knowledge distillation on the smaller variants."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.00118/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":129,"sample":[{"doi":"","year":2024,"title":"R. Agarwal, N. Vieillard, Y. Zhou, P. Stanczyk, S. R. Garea, M. Geist, and O. Bachem. On-policy distillation of language models: Learning from self-generated mistakes. In The Twelfth International Con","work_id":"a7703167-db43-41f4-a282-4cdc8c584424","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2024,"title":"Llama 3 model card","work_id":"008d23c2-07d0-4784-a704-56521d627c6b","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"E. Almazrouei, H. Alobeidli, A. Alshamsi, A. Cappelli, R. Cojocaru, M. Debbah, Étienne Goffinet, D. Hesslow, J. Launay, Q. Malartic, D. Mazzotta, B. Noune, B. Pannier, and G. Penedo. The falcon series","work_id":"29ce5015-6493-4cdb-9d40-83551c1d7058","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2022,"title":"P. Barham, A. Chowdhery, J. Dean, S. Ghemawat, S. Hand, D. Hurt, M. Isard, H. Lim, R. Pang, S. Roy, B. Saeta, P. Schuh, R. Sepassi, L. E. Shafey, C. A. Thekkath, and Y. Wu. Pathways: Asynchronous dist","work_id":"d3d67d96-f15e-42d9-9361-236a63e80984","ref_index":8,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2024,"title":"W.-L. Chiang, L. Zheng, Y. Sheng, A. N. Angelopoulos, T. Li, D. Li, H. Zhang, B. Zhu, M. Jordan, J. E. Gonzalez, and I. Stoica. Chatbot arena: An open platform for evaluating llms by human preference,","work_id":"0a2e0bf3-b2de-4c73-9707-690b4d5f6246","ref_index":15,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":129,"snapshot_sha256":"ae4f9c15fd226d5ab9c65ad0791b7ebc501d0e377a8a024a836817b29c57de7c","internal_anchors":31},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"305abfa3-83f3-4c36-8445-cb015f283e1f"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:14:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"n3woi1tbxk9rgt/vTiTOd8g3CLeeeOw+30eTYtlGVzMuBC3cb+vKdTrGdw/AAxKtWh7r7CX0HzMf6NfpiYNBBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T00:23:06.279837Z"},"content_sha256":"12fe6ad92dd112c75c6c045bf29b4f1916c0d5183e697d98ae7b7edbff7a158d","schema_version":"1.0","event_id":"sha256:12fe6ad92dd112c75c6c045bf29b4f1916c0d5183e697d98ae7b7edbff7a158d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2LR4LPBOWIE25TP3CYFS33XNSD/bundle.json","state_url":"https://pith.science/pith/2LR4LPBOWIE25TP3CYFS33XNSD/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2LR4LPBOWIE25TP3CYFS33XNSD/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T00:23:06Z","links":{"resolver":"https://pith.science/pith/2LR4LPBOWIE25TP3CYFS33XNSD","bundle":"https://pith.science/pith/2LR4LPBOWIE25TP3CYFS33XNSD/bundle.json","state":"https://pith.science/pith/2LR4LPBOWIE25TP3CYFS33XNSD/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2LR4LPBOWIE25TP3CYFS33XNSD/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:2LR4LPBOWIE25TP3CYFS33XNSD","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"988d94bd4b98ba8eb221303a3c8a478ebeb329bc038811a13566339403e8f79b","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-31T19:13:07Z","title_canon_sha256":"f0c83b1d378b9b8629c78b542bb35f200a1e613dfaa0029096f206b3af1fd760"},"schema_version":"1.0","source":{"id":"2408.00118","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2408.00118","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"arxiv_version","alias_value":"2408.00118v3","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.00118","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"pith_short_12","alias_value":"2LR4LPBOWIE2","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"pith_short_16","alias_value":"2LR4LPBOWIE25TP3","created_at":"2026-07-05T09:14:36Z"},{"alias_kind":"pith_short_8","alias_value":"2LR4LPBO","created_at":"2026-07-05T09:14:36Z"}],"graph_snapshots":[{"event_id":"sha256:12fe6ad92dd112c75c6c045bf29b4f1916c0d5183e697d98ae7b7edbff7a158d","target":"graph","created_at":"2026-07-05T09:14:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"The resulting models deliver the best performance for their size, and even offer competitive alternatives to models that are 2-3 times bigger."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the reported performance gains are attributable to the listed architectural changes and distillation rather than to undisclosed differences in training data, compute, or evaluation protocols."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Gemma 2 models achieve leading performance at their sizes by combining established Transformer modifications with knowledge distillation for the 2B and 9B variants."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Gemma 2 models achieve leading performance at their sizes through interleaving local-global attention, group-query attention, and knowledge distillation on the smaller variants."}],"snapshot_sha256":"f1176a06c6a324417e3b4d343bd1fc732f2f2c3a07552d7c91fb9f2ed3c972a7"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2408.00118/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this work, we introduce Gemma 2, a new addition to the Gemma family of lightweight, state-of-the-art open models, ranging in scale from 2 billion to 27 billion parameters. In this new version, we apply several known technical modifications to the Transformer architecture, such as interleaving local-global attentions (Beltagy et al., 2020a) and group-query attention (Ainslie et al., 2023). We also train the 2B and 9B models with knowledge distillation (Hinton et al., 2015) instead of next token prediction. The resulting models deliver the best performance for their size, and even offer compe","authors_text":"Abe Friesen, Alanna Walton, Alek Andreev, Alexandre Ram\\'e, Aliaksei Severyn, Alicia Parrish, Aliya Ahmad, Allen Hutchison, Alvin Abdagic, Amanda Carl, Amy Shen, Anand Rao, Anca Dragan, Andy Brock, Andy Coenen, Anthony Laforge, Antonia Paterson, Anton Tsitsulin, Armand Joulin, Behnam Neyshabur, Ben Bastian, Bilal Piot, Bobak Shahriari, Bo Wu, Brandon Royal, Cassidy Hardin, Charlie Chen, Charline Le Lan, Chintu Kumar, Chris Perry, Christopher A. Choquette-Choo, Chris Welty, Clement Farabet, Danila Sinopalnikov, David Weinberger, Demis Hassabis, Dimple Vijaykumar, Dominika Rogozi\\'nska, D. Sculley, Dustin Herbison, Elena Buchatskaya, Eli Collins, Elisa Bandy, Emma Wang, Erica Moreira, Eric Noland, Evan Senter, Evgenii Eltyshev, Francesco Visin, Gabriel Rasskin, Gary Wei, Gemma Team: Morgane Riviere, Glenn Cameron, Gus Martins, Hadi Hashemi, Hanna Klimczak-Pluci\\'nska, Harleen Batra, Harsh Dhand, Ivan Nardini, Jacinda Mein, Jack Zhou, James Svensson, Jean-bastien Grill, Jeanine Banks, Jeff Dean, Jeff Stanway, Jetha Chan, Jin Peng Zhou, Joana Carrasqueira, Joana Iljazi, Jocelyn Becker, Joe Fernandez, Joelle Barral, Johan Ferret, Joost van Amersfoort, Josh Gordon, Josh Lipschultz, Josh Newlan, Ju-yeong Ji, Kareem Mohamed, Kartikeya Badola, Kat Black, Kathleen Kenealy, Katie Millican, Keelin McDonell, Kelvin Nguyen, Kiranbir Sodhia, Kish Greene, Koray Kavukcuoglu, Lars Lowe Sjoesund, Laurent Sifre, Lauren Usui, Lena Heuermann, L\\'eonard Hussenot, Leticia Lago, Lilly McNealus, Livio Baldini Soares, Logan Kilpatrick, Lucas Dixon, Luciano Martins, Ludovic Peran, Machel Reid, Manvinder Singh, Mark Iverson, Martin G\\\"orner, Mateo Wirth, Matt Davidow, Matthew Rahtz, Matthew Watson, Matt Hoffman, Matt Miller, Mat Velloso, Meg Risdal, Mehran Kazemi, Michael Moynihan, Michelle Casbon, Ming Zhang, Minh Giang, Minsuk Kahng, Minwoo Park, Mofi Rahman, Mohit Khatwani, Natalie Dao, Nenshad Bardoliwalla, Nesh Devanathan, Neta Dumai, Nikola Momchev, Nilay Chauhan, Nino Vieillard, Noah Fiedel, Olivier Bachem, Oriol Vinyals, Oscar Wahltinez, Pankil Botarda, Parker Barnes, Paul Barham, Paul Michel, Pengchong Jin, Peter Liu, Petko Georgiev, Phil Culliton, Phoebe Kirk, Pier Giuseppe Sessa, Piotr Stanczyk, Pouya Tafti, Pradeep Kuppala, Raia Hadsell, Ramona Comanescu, Ramona Merhej, Ravin Kumar, Reena Jana, Reza Ardeshir Rokni, Rishabh Agarwal, Robert Dadashi, Ryan Mullins, Sabela Ramos, Samaneh Saadat, Sammy Jerome, Sarah Cogan, Sarah Perrin, Sara Mc Carthy, Sebastian Borgeaud, Sebastian Krause, S\\'ebastien M. R. Arnold, Sertan Girgin, Shantanu Thakoor, Shengyang Dai, Shreya Pathak, Shruti Garg, Shruti Sheth, Slav Petrov, Sue Ronstrom, Surya Bhupatiraju, Susan Chan, Thomas Mesnard, Timothy Jordan, Ting Yu, Tomas Kocisky, Tom Eccles, Tom Hennigan, Tris Warkentin, Tulsee Doshi, Victor Cotruta, Vihan Jain, Vikas Yadav, Vilobh Meshram, Vishal Dharmadhikari, Warren Barkley, Wei Wei, Wenming Ye, Woohyun Han, Woosuk Kwon, Xiang Xu, Zhe Shen, Zhitao Gong, Zichuan Wei, Zoubin Ghahramani","cross_cats":["cs.AI"],"headline":"Gemma 2 models achieve leading performance at their sizes through interleaving local-global attention, group-query attention, and knowledge distillation on the smaller variants.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size"},"references":{"count":129,"internal_anchors":31,"resolved_work":129,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"R. Agarwal, N. Vieillard, Y. Zhou, P. Stanczyk, S. R. Garea, M. Geist, and O. Bachem. On-policy distillation of language models: Learning from self-generated mistakes. In The Twelfth International Con","work_id":"a7703167-db43-41f4-a282-4cdc8c584424","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Llama 3 model card","work_id":"008d23c2-07d0-4784-a704-56521d627c6b","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"E. Almazrouei, H. Alobeidli, A. Alshamsi, A. Cappelli, R. Cojocaru, M. Debbah, Étienne Goffinet, D. Hesslow, J. Launay, Q. Malartic, D. Mazzotta, B. Noune, B. Pannier, and G. Penedo. The falcon series","work_id":"29ce5015-6493-4cdb-9d40-83551c1d7058","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":8,"title":"P. Barham, A. Chowdhery, J. Dean, S. Ghemawat, S. Hand, D. Hurt, M. Isard, H. Lim, R. Pang, S. Roy, B. Saeta, P. Schuh, R. Sepassi, L. E. Shafey, C. A. Thekkath, and Y. Wu. Pathways: Asynchronous dist","work_id":"d3d67d96-f15e-42d9-9361-236a63e80984","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":15,"title":"W.-L. Chiang, L. Zheng, Y. Sheng, A. N. Angelopoulos, T. Li, D. Li, H. Zhang, B. Zhu, M. Jordan, J. E. Gonzalez, and I. Stoica. Chatbot arena: An open platform for evaluating llms by human preference,","work_id":"0a2e0bf3-b2de-4c73-9707-690b4d5f6246","year":2024}],"snapshot_sha256":"ae4f9c15fd226d5ab9c65ad0791b7ebc501d0e377a8a024a836817b29c57de7c"},"source":{"id":"2408.00118","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-10T12:06:11.309894Z","id":"305abfa3-83f3-4c36-8445-cb015f283e1f","model_set":{"reader":"grok-4.3"},"one_line_summary":"Gemma 2 models achieve leading performance at their sizes by combining established Transformer modifications with knowledge distillation for the 2B and 9B variants.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Gemma 2 models achieve leading performance at their sizes through interleaving local-global attention, group-query attention, and knowledge distillation on the smaller variants.","strongest_claim":"The resulting models deliver the best performance for their size, and even offer competitive alternatives to models that are 2-3 times bigger.","weakest_assumption":"That the reported performance gains are attributable to the listed architectural changes and distillation rather than to undisclosed differences in training data, compute, or evaluation protocols."}},"verdict_id":"305abfa3-83f3-4c36-8445-cb015f283e1f"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:007f9290634d32cff2edb01ca4fdae81219f9148ba88caf4cc587bde798b5107","target":"record","created_at":"2026-07-05T09:14:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"988d94bd4b98ba8eb221303a3c8a478ebeb329bc038811a13566339403e8f79b","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-31T19:13:07Z","title_canon_sha256":"f0c83b1d378b9b8629c78b542bb35f200a1e613dfaa0029096f206b3af1fd760"},"schema_version":"1.0","source":{"id":"2408.00118","kind":"arxiv","version":3}},"canonical_sha256":"d2e3c5bc2eb209aecdfb160b2deeed90ff69a48a5f7c698d2e51441375eec723","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d2e3c5bc2eb209aecdfb160b2deeed90ff69a48a5f7c698d2e51441375eec723","first_computed_at":"2026-07-05T09:14:36.990462Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:14:36.990462Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oTKuLa6oZ+kf1Ytg/UrYeeQzw1Wyr5kP8S+T3QXg9FkR24SK9a1qCBih31Hhh+9md5+juDtSmOP3mrPyJxnyCg==","signature_status":"signed_v1","signed_at":"2026-07-05T09:14:36.990964Z","signed_message":"canonical_sha256_bytes"},"source_id":"2408.00118","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:007f9290634d32cff2edb01ca4fdae81219f9148ba88caf4cc587bde798b5107","sha256:12fe6ad92dd112c75c6c045bf29b4f1916c0d5183e697d98ae7b7edbff7a158d"],"state_sha256":"4d579037becd5d140e9aac30774009adec4e91c68cafe125b804b9622207a6dc"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xf9zS9cdCfErORtUKa0Cc+KTTgPIbaxiDunLv8dHyXbrQSqMZxB98zDAYLO4iv8Mpnnbo+cAz64egkHUXN/cDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T00:23:06.284339Z","bundle_sha256":"d60cdf588f410de7e0731a8ca49da7c0a45127aa7605eb978c6d56b5e572116f"}}