{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DKVYV7KCQNALQXPE3E2TY2GZZV","short_pith_number":"pith:DKVYV7KC","schema_version":"1.0","canonical_sha256":"1aab8afd428340b85de4d9353c68d9cd499247aa77e35b385bfbd752207ebacf","source":{"kind":"arxiv","id":"2408.15417","version":2},"attestation_state":"computed","paper":{"title":"Implicit Geometry of Next-token Prediction: From Language Sparsity Patterns to Model Representations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christos Thrampoulidis, Tina Behnia, Vala Vakilian, Yize Zhao","submitted_at":"2024-08-27T21:46:47Z","abstract_excerpt":"Next-token prediction (NTP) over large text corpora has become the go-to paradigm to train large language models. Yet, it remains unclear how NTP influences the mapping of linguistic patterns to geometric properties of the resulting model representations. We frame training of large language models as soft-label classification over sparse probabilistic label vectors, coupled with an analytical approximation that allows unrestricted generation of context embeddings. This approach links NTP training to rank-constrained, nuclear-norm regularized optimization in the logit domain, offering a framewo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.15417","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-27T21:46:47Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b1ff173b1e1e1d23bfb14ce9daacd593c2e699bc3dede5c2728b2392c1358d62","abstract_canon_sha256":"c68a60336b6a57c41d16738092703a7f132e04cee17e045bfd5dfb74b2a35cef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:37.626518Z","signature_b64":"bKcUWE/PA7lS0KiVJAYTSoD5Yb99JU7lGsl2mNiccsoy76qKezcjU+AN5c/gaKbP5q0TLI7G51c/4MotcxYqDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1aab8afd428340b85de4d9353c68d9cd499247aa77e35b385bfbd752207ebacf","last_reissued_at":"2026-07-05T10:16:37.625823Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:37.625823Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Implicit Geometry of Next-token Prediction: From Language Sparsity Patterns to Model Representations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Christos Thrampoulidis, Tina Behnia, Vala Vakilian, Yize Zhao","submitted_at":"2024-08-27T21:46:47Z","abstract_excerpt":"Next-token prediction (NTP) over large text corpora has become the go-to paradigm to train large language models. Yet, it remains unclear how NTP influences the mapping of linguistic patterns to geometric properties of the resulting model representations. We frame training of large language models as soft-label classification over sparse probabilistic label vectors, coupled with an analytical approximation that allows unrestricted generation of context embeddings. This approach links NTP training to rank-constrained, nuclear-norm regularized optimization in the logit domain, offering a framewo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.15417","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.15417/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.15417","created_at":"2026-07-05T10:16:37.625926+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.15417v2","created_at":"2026-07-05T10:16:37.625926+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.15417","created_at":"2026-07-05T10:16:37.625926+00:00"},{"alias_kind":"pith_short_12","alias_value":"DKVYV7KCQNAL","created_at":"2026-07-05T10:16:37.625926+00:00"},{"alias_kind":"pith_short_16","alias_value":"DKVYV7KCQNALQXPE","created_at":"2026-07-05T10:16:37.625926+00:00"},{"alias_kind":"pith_short_8","alias_value":"DKVYV7KC","created_at":"2026-07-05T10:16:37.625926+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.26745","citing_title":"Deep sequence models tend to memorize geometrically; it is unclear why","ref_index":206,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17084","citing_title":"Scale Determines Whether Language Models Organize Representation Geometry for Prediction","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV","json":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV.json","graph_json":"https://pith.science/api/pith-number/DKVYV7KCQNALQXPE3E2TY2GZZV/graph.json","events_json":"https://pith.science/api/pith-number/DKVYV7KCQNALQXPE3E2TY2GZZV/events.json","paper":"https://pith.science/paper/DKVYV7KC"},"agent_actions":{"view_html":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV","download_json":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV.json","view_paper":"https://pith.science/paper/DKVYV7KC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.15417&json=true","fetch_graph":"https://pith.science/api/pith-number/DKVYV7KCQNALQXPE3E2TY2GZZV/graph.json","fetch_events":"https://pith.science/api/pith-number/DKVYV7KCQNALQXPE3E2TY2GZZV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV/action/storage_attestation","attest_author":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV/action/author_attestation","sign_citation":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV/action/citation_signature","submit_replication":"https://pith.science/pith/DKVYV7KCQNALQXPE3E2TY2GZZV/action/replication_record"}},"created_at":"2026-07-05T10:16:37.625926+00:00","updated_at":"2026-07-05T10:16:37.625926+00:00"}