{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GCZ2TABVTX2NY2QEVHH2VKPEQK","short_pith_number":"pith:GCZ2TABV","schema_version":"1.0","canonical_sha256":"30b3a980359df4dc6a04a9cfaaa9e48280b05457d99e02ada51696f0b8b2a381","source":{"kind":"arxiv","id":"2407.10827","version":2},"attestation_state":"computed","paper":{"title":"LLM Circuit Analyses Are Consistent Across Training and Scale","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Curt Tigges, Michael Hanna, Qinan Yu, Stella Biderman","submitted_at":"2024-07-15T15:38:51Z","abstract_excerpt":"Most currently deployed large language models (LLMs) undergo continuous training or additional finetuning. By contrast, most research into LLMs' internal mechanisms focuses on models at one snapshot in time (the end of pre-training), raising the question of whether their results generalize to real-world settings. Existing studies of mechanisms over time focus on encoder-only or toy models, which differ significantly from most deployed models. In this study, we track how model mechanisms, operationalized as circuits, emerge and evolve across 300 billion tokens of training in decoder-only LLMs, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10827","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-15T15:38:51Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"fdcc30020bf0c3fed7a166394ceb77d63e2f25fb5e8198da3a0235c29af84892","abstract_canon_sha256":"9c5744b0812b65aa03f73515d6db58eb534e0a5ea4a10cfe0505d967915f2c7b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:40:18.636806Z","signature_b64":"5DmXo0RmO3EzIfcz1espCGYlONaIbR553T5dx6d95VeBgSMntD6DndVo93xWkdp157dM3jE65ChjFc9XRavuDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30b3a980359df4dc6a04a9cfaaa9e48280b05457d99e02ada51696f0b8b2a381","last_reissued_at":"2026-07-05T09:40:18.636086Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:40:18.636086Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM Circuit Analyses Are Consistent Across Training and Scale","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Curt Tigges, Michael Hanna, Qinan Yu, Stella Biderman","submitted_at":"2024-07-15T15:38:51Z","abstract_excerpt":"Most currently deployed large language models (LLMs) undergo continuous training or additional finetuning. By contrast, most research into LLMs' internal mechanisms focuses on models at one snapshot in time (the end of pre-training), raising the question of whether their results generalize to real-world settings. Existing studies of mechanisms over time focus on encoder-only or toy models, which differ significantly from most deployed models. In this study, we track how model mechanisms, operationalized as circuits, emerge and evolve across 300 billion tokens of training in decoder-only LLMs, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10827","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10827","created_at":"2026-07-05T09:40:18.636189+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10827v2","created_at":"2026-07-05T09:40:18.636189+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10827","created_at":"2026-07-05T09:40:18.636189+00:00"},{"alias_kind":"pith_short_12","alias_value":"GCZ2TABVTX2N","created_at":"2026-07-05T09:40:18.636189+00:00"},{"alias_kind":"pith_short_16","alias_value":"GCZ2TABVTX2NY2QE","created_at":"2026-07-05T09:40:18.636189+00:00"},{"alias_kind":"pith_short_8","alias_value":"GCZ2TABV","created_at":"2026-07-05T09:40:18.636189+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06621","citing_title":"Fingerprint, Not Blueprint: How Positional Schemes Set the Default Spectral Algebra of Attention","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK","json":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK.json","graph_json":"https://pith.science/api/pith-number/GCZ2TABVTX2NY2QEVHH2VKPEQK/graph.json","events_json":"https://pith.science/api/pith-number/GCZ2TABVTX2NY2QEVHH2VKPEQK/events.json","paper":"https://pith.science/paper/GCZ2TABV"},"agent_actions":{"view_html":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK","download_json":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK.json","view_paper":"https://pith.science/paper/GCZ2TABV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10827&json=true","fetch_graph":"https://pith.science/api/pith-number/GCZ2TABVTX2NY2QEVHH2VKPEQK/graph.json","fetch_events":"https://pith.science/api/pith-number/GCZ2TABVTX2NY2QEVHH2VKPEQK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK/action/storage_attestation","attest_author":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK/action/author_attestation","sign_citation":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK/action/citation_signature","submit_replication":"https://pith.science/pith/GCZ2TABVTX2NY2QEVHH2VKPEQK/action/replication_record"}},"created_at":"2026-07-05T09:40:18.636189+00:00","updated_at":"2026-07-05T09:40:18.636189+00:00"}