{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7HO6W5XTSYCXQCNSLU2IDTSKZD","short_pith_number":"pith:7HO6W5XT","schema_version":"1.0","canonical_sha256":"f9ddeb76f396057809b25d3481ce4ac8f4bd7e411f27bfb14204a684a407539b","source":{"kind":"arxiv","id":"2404.14462","version":4},"attestation_state":"computed","paper":{"title":"Towards smaller, faster decoder-only transformers: Architectural variants and their implications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Sathya Krishnan Suresh, Shunmugapriya P","submitted_at":"2024-04-22T06:19:46Z","abstract_excerpt":"In recent times, the research on Large Language Models (LLMs) has grown exponentially, predominantly focusing on models underpinned by the transformer architecture, as established by [1], and further developed through the decoder-only variations by [2]. Contemporary efforts in this field primarily aim to enhance model capabilities by scaling up both the architecture and data volumes utilized during training. However, the exploration into reduce these model sizes while preserving their efficacy remains scant. In this study, we introduce three modifications to the decoder-only transformer archit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14462","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-22T06:19:46Z","cross_cats_sorted":[],"title_canon_sha256":"1d39c39b8dab3f307774143b992531bca194a10016f4b8a3e57fe42a15584e95","abstract_canon_sha256":"db4dbbb1ec88b8d38428e4c22e621df0d3784667c767060e1dd3c3a28bc26fc6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:40.304731Z","signature_b64":"IpIgXXMFXRCYP6JfTHgQnr9SIRaftC+9r297nkxCev375fMm779vZbd6lJDNn1g4frXt1idRbBJBcw7bfTLsDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9ddeb76f396057809b25d3481ce4ac8f4bd7e411f27bfb14204a684a407539b","last_reissued_at":"2026-07-05T09:17:40.304223Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:40.304223Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards smaller, faster decoder-only transformers: Architectural variants and their implications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Sathya Krishnan Suresh, Shunmugapriya P","submitted_at":"2024-04-22T06:19:46Z","abstract_excerpt":"In recent times, the research on Large Language Models (LLMs) has grown exponentially, predominantly focusing on models underpinned by the transformer architecture, as established by [1], and further developed through the decoder-only variations by [2]. Contemporary efforts in this field primarily aim to enhance model capabilities by scaling up both the architecture and data volumes utilized during training. However, the exploration into reduce these model sizes while preserving their efficacy remains scant. In this study, we introduce three modifications to the decoder-only transformer archit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14462","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14462/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14462","created_at":"2026-07-05T09:17:40.304282+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14462v4","created_at":"2026-07-05T09:17:40.304282+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14462","created_at":"2026-07-05T09:17:40.304282+00:00"},{"alias_kind":"pith_short_12","alias_value":"7HO6W5XTSYCX","created_at":"2026-07-05T09:17:40.304282+00:00"},{"alias_kind":"pith_short_16","alias_value":"7HO6W5XTSYCXQCNS","created_at":"2026-07-05T09:17:40.304282+00:00"},{"alias_kind":"pith_short_8","alias_value":"7HO6W5XT","created_at":"2026-07-05T09:17:40.304282+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.05028","citing_title":"Evaluation of Finetuned LLMs in AMR Parsing","ref_index":57,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD","json":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD.json","graph_json":"https://pith.science/api/pith-number/7HO6W5XTSYCXQCNSLU2IDTSKZD/graph.json","events_json":"https://pith.science/api/pith-number/7HO6W5XTSYCXQCNSLU2IDTSKZD/events.json","paper":"https://pith.science/paper/7HO6W5XT"},"agent_actions":{"view_html":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD","download_json":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD.json","view_paper":"https://pith.science/paper/7HO6W5XT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14462&json=true","fetch_graph":"https://pith.science/api/pith-number/7HO6W5XTSYCXQCNSLU2IDTSKZD/graph.json","fetch_events":"https://pith.science/api/pith-number/7HO6W5XTSYCXQCNSLU2IDTSKZD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD/action/storage_attestation","attest_author":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD/action/author_attestation","sign_citation":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD/action/citation_signature","submit_replication":"https://pith.science/pith/7HO6W5XTSYCXQCNSLU2IDTSKZD/action/replication_record"}},"created_at":"2026-07-05T09:17:40.304282+00:00","updated_at":"2026-07-05T09:17:40.304282+00:00"}