{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YJCJARO5FJAK7XS3PPW2VBTR24","short_pith_number":"pith:YJCJARO5","schema_version":"1.0","canonical_sha256":"c2449045dd2a40afde5b7bedaa8671d73c178ef88e7ae855d1a467659c2f4be9","source":{"kind":"arxiv","id":"2407.12034","version":2},"attestation_state":"computed","paper":{"title":"Understanding Transformers via N-gram Statistics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Timothy Nguyen","submitted_at":"2024-06-30T22:18:49Z","abstract_excerpt":"Transformer based large-language models (LLMs) display extreme proficiency with language yet a precise understanding of how they work remains elusive. One way of demystifying transformer predictions would be to describe how they depend on their context in terms of simple template functions. This paper takes a first step in this direction by considering families of functions (i.e. rules) formed out of simple N-gram based statistics of the training data. By studying how well these rulesets approximate transformer predictions, we obtain a variety of novel discoveries: a simple method to detect ov"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12034","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-30T22:18:49Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"4df864e126d4eaeb05e085aac583c70bb1b9a0afd4c3b2f4f44385bdccc550d3","abstract_canon_sha256":"ccb225c8fd897630fc399e4f3c972221c7d7dc9bc0e41e9973e2bc2b5bcfb4e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:05.807236Z","signature_b64":"XGgIrmZ7vEmmaqZl4Hm8Hjk7zufMUp+vY7HSrucPcK91WVJrNIVWm/ej3qwt98sbvSUUkbAeHDdAwGxd+pRLDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2449045dd2a40afde5b7bedaa8671d73c178ef88e7ae855d1a467659c2f4be9","last_reissued_at":"2026-07-05T09:31:05.806786Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:05.806786Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Transformers via N-gram Statistics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Timothy Nguyen","submitted_at":"2024-06-30T22:18:49Z","abstract_excerpt":"Transformer based large-language models (LLMs) display extreme proficiency with language yet a precise understanding of how they work remains elusive. One way of demystifying transformer predictions would be to describe how they depend on their context in terms of simple template functions. This paper takes a first step in this direction by considering families of functions (i.e. rules) formed out of simple N-gram based statistics of the training data. By studying how well these rulesets approximate transformer predictions, we obtain a variety of novel discoveries: a simple method to detect ov"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12034","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12034/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12034","created_at":"2026-07-05T09:31:05.806851+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12034v2","created_at":"2026-07-05T09:31:05.806851+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12034","created_at":"2026-07-05T09:31:05.806851+00:00"},{"alias_kind":"pith_short_12","alias_value":"YJCJARO5FJAK","created_at":"2026-07-05T09:31:05.806851+00:00"},{"alias_kind":"pith_short_16","alias_value":"YJCJARO5FJAK7XS3","created_at":"2026-07-05T09:31:05.806851+00:00"},{"alias_kind":"pith_short_8","alias_value":"YJCJARO5","created_at":"2026-07-05T09:31:05.806851+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15183","citing_title":"When Are Two Networks the Same? Tensor Similarity for Mechanistic Interpretability","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01381","citing_title":"A framework for analyzing concept representations in neural models","ref_index":151,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24","json":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24.json","graph_json":"https://pith.science/api/pith-number/YJCJARO5FJAK7XS3PPW2VBTR24/graph.json","events_json":"https://pith.science/api/pith-number/YJCJARO5FJAK7XS3PPW2VBTR24/events.json","paper":"https://pith.science/paper/YJCJARO5"},"agent_actions":{"view_html":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24","download_json":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24.json","view_paper":"https://pith.science/paper/YJCJARO5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12034&json=true","fetch_graph":"https://pith.science/api/pith-number/YJCJARO5FJAK7XS3PPW2VBTR24/graph.json","fetch_events":"https://pith.science/api/pith-number/YJCJARO5FJAK7XS3PPW2VBTR24/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24/action/storage_attestation","attest_author":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24/action/author_attestation","sign_citation":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24/action/citation_signature","submit_replication":"https://pith.science/pith/YJCJARO5FJAK7XS3PPW2VBTR24/action/replication_record"}},"created_at":"2026-07-05T09:31:05.806851+00:00","updated_at":"2026-07-05T09:31:05.806851+00:00"}