{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KX3IH3DUBLYYOPRDE76YS2KJ2J","short_pith_number":"pith:KX3IH3DU","schema_version":"1.0","canonical_sha256":"55f683ec740af1873e2327fd896949d242a005f161eb2091371685f9ecb3a15c","source":{"kind":"arxiv","id":"2405.14159","version":2},"attestation_state":"computed","paper":{"title":"Super Tiny Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bobby Cheng, Chen Ruirui, Cheston Tan, Dylan Hillier, Leon Guertler, Palaash Agrawal","submitted_at":"2024-05-23T04:12:49Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has led to significant improvements in natural language processing but also poses challenges due to their high computational and energy demands. This paper introduces a series of research efforts focused on Super Tiny Language Models (STLMs), which aim to deliver high performance with significantly reduced parameter counts. We explore innovative techniques such as byte-level tokenization with a pooling mechanism, weight tying, and efficient training strategies. These methods aim to significantly reduce reduce the parameter count compared to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14159","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-23T04:12:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e48da8a352604f39fc382e35a7cc22f5d48724da82702367f08021d8108e4ddc","abstract_canon_sha256":"5d27b407be424796484b8a8555092dbbb90cf891da5aa5306aa8ee4c5bd31780"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:47.866442Z","signature_b64":"uVYSOBIIOxM3AjuY1bKwCY+UnkjkMGLyXI2cfp7BqPMfyK2mmU0ZH+pcrXmDBD18l9Kv7XIviWIyIW4mfKC+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"55f683ec740af1873e2327fd896949d242a005f161eb2091371685f9ecb3a15c","last_reissued_at":"2026-07-05T08:36:47.865951Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:47.865951Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Super Tiny Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bobby Cheng, Chen Ruirui, Cheston Tan, Dylan Hillier, Leon Guertler, Palaash Agrawal","submitted_at":"2024-05-23T04:12:49Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has led to significant improvements in natural language processing but also poses challenges due to their high computational and energy demands. This paper introduces a series of research efforts focused on Super Tiny Language Models (STLMs), which aim to deliver high performance with significantly reduced parameter counts. We explore innovative techniques such as byte-level tokenization with a pooling mechanism, weight tying, and efficient training strategies. These methods aim to significantly reduce reduce the parameter count compared to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14159","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14159/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14159","created_at":"2026-07-05T08:36:47.866008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14159v2","created_at":"2026-07-05T08:36:47.866008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14159","created_at":"2026-07-05T08:36:47.866008+00:00"},{"alias_kind":"pith_short_12","alias_value":"KX3IH3DUBLYY","created_at":"2026-07-05T08:36:47.866008+00:00"},{"alias_kind":"pith_short_16","alias_value":"KX3IH3DUBLYYOPRD","created_at":"2026-07-05T08:36:47.866008+00:00"},{"alias_kind":"pith_short_8","alias_value":"KX3IH3DU","created_at":"2026-07-05T08:36:47.866008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.15461","citing_title":"All is Not Lost: LLM Recovery without Checkpoints","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15461","citing_title":"All is Not Lost: LLM Recovery without Checkpoints","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16037","citing_title":"Stochasticity in Tokenisation Improves Robustness","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J","json":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J.json","graph_json":"https://pith.science/api/pith-number/KX3IH3DUBLYYOPRDE76YS2KJ2J/graph.json","events_json":"https://pith.science/api/pith-number/KX3IH3DUBLYYOPRDE76YS2KJ2J/events.json","paper":"https://pith.science/paper/KX3IH3DU"},"agent_actions":{"view_html":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J","download_json":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J.json","view_paper":"https://pith.science/paper/KX3IH3DU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14159&json=true","fetch_graph":"https://pith.science/api/pith-number/KX3IH3DUBLYYOPRDE76YS2KJ2J/graph.json","fetch_events":"https://pith.science/api/pith-number/KX3IH3DUBLYYOPRDE76YS2KJ2J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J/action/storage_attestation","attest_author":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J/action/author_attestation","sign_citation":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J/action/citation_signature","submit_replication":"https://pith.science/pith/KX3IH3DUBLYYOPRDE76YS2KJ2J/action/replication_record"}},"created_at":"2026-07-05T08:36:47.866008+00:00","updated_at":"2026-07-05T08:36:47.866008+00:00"}