{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:N5HVAVYLAFC56L7NWBRK36RJWX","short_pith_number":"pith:N5HVAVYL","schema_version":"1.0","canonical_sha256":"6f4f50570b0145df2fedb062adfa29b5f542ee09bb77ec15449b6bb1ad96cf62","source":{"kind":"arxiv","id":"2310.02207","version":3},"attestation_state":"computed","paper":{"title":"Language Models Represent Space and Time","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Max Tegmark, Wes Gurnee","submitted_at":"2023-10-03T17:06:52Z","abstract_excerpt":"The capabilities of large language models (LLMs) have sparked debate over whether such systems just learn an enormous collection of superficial statistics or a set of more coherent and grounded representations that reflect the real world. We find evidence for the latter by analyzing the learned representations of three spatial datasets (world, US, NYC places) and three temporal datasets (historical figures, artworks, news headlines) in the Llama-2 family of models. We discover that LLMs learn linear representations of space and time across multiple scales. These representations are robust to p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.02207","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-03T17:06:52Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"28e4a950a990b8fb847e5adb498ce81fda88e751981d6f4f22ecec700262fb4f","abstract_canon_sha256":"fdf392c2afec35a9552d46cae7e1757cda685419f97be8826e72c144d8a6af1b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:33.115843Z","signature_b64":"DwYZQmSDxuuyh3g5JFIGedW/KrK1X6GdPQixibwd3XEopclD9AIWy/3tYEL6V+LRGOmDhjdX+WHh9Yw/IybTCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f4f50570b0145df2fedb062adfa29b5f542ee09bb77ec15449b6bb1ad96cf62","last_reissued_at":"2026-07-05T07:51:33.115254Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:33.115254Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Models Represent Space and Time","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Max Tegmark, Wes Gurnee","submitted_at":"2023-10-03T17:06:52Z","abstract_excerpt":"The capabilities of large language models (LLMs) have sparked debate over whether such systems just learn an enormous collection of superficial statistics or a set of more coherent and grounded representations that reflect the real world. We find evidence for the latter by analyzing the learned representations of three spatial datasets (world, US, NYC places) and three temporal datasets (historical figures, artworks, news headlines) in the Llama-2 family of models. We discover that LLMs learn linear representations of space and time across multiple scales. These representations are robust to p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.02207","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.02207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.02207","created_at":"2026-07-05T07:51:33.115340+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.02207v3","created_at":"2026-07-05T07:51:33.115340+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.02207","created_at":"2026-07-05T07:51:33.115340+00:00"},{"alias_kind":"pith_short_12","alias_value":"N5HVAVYLAFC5","created_at":"2026-07-05T07:51:33.115340+00:00"},{"alias_kind":"pith_short_16","alias_value":"N5HVAVYLAFC56L7N","created_at":"2026-07-05T07:51:33.115340+00:00"},{"alias_kind":"pith_short_8","alias_value":"N5HVAVYL","created_at":"2026-07-05T07:51:33.115340+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07047","citing_title":"Riemannian Geometry for Pre-trained Language Model Embeddings","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24669","citing_title":"LaGO: Latent Action Guidance for Online Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21345","citing_title":"Factual Retrieval in LLMs Is a Redundant, Distributed and Non-Contiguous Process","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20152","citing_title":"From Texts to Scores: Tracing the Emergence of Essay Quality Representations in Large Language Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12689","citing_title":"Observable Patterns Are Not Explanations: A Causal-Geometric Analysis of Latent Reasoning Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09658","citing_title":"Muon Learns More Robust and Transferable Features than Adam","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04381","citing_title":"From Symbolic to Geometric: Enabling Spatial Reasoning in Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03022","citing_title":"Hallucinations as Orthogonal Noise: Inference-Time Manifold Alignment via Dynamic Contextual Orthogonalization","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03685","citing_title":"A Close Look At World Model Recovery In Supervised Fine-Tuned LLM Planners","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31404","citing_title":"The Sword, Shield, and Achilles' Heel: Characterizing the Linguistic Inductive Bias of Large Language Models for Spatial Reasoning in Navigation Planning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28548","citing_title":"Turn-Averaged SAEs for Feature Discovery and Long-Context Attribution","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20603","citing_title":"A Survey of Large Language Models for Perception and Measurement of Human Psychology","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07568","citing_title":"A Systematic Study of Behavioral Cloning for Scientific Data Annotation","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27970","citing_title":"Geometry of Human Perceptual Domains Emerges Transiently in LLM Representations","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29358","citing_title":"Scaling Monosemanticity: Extracting Interpretable Features from Claude 3 Sonnet","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20761","citing_title":"Integrating Large Language Model Agents with Digital Twins for Industrial Autonomous Systems","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04952","citing_title":"Quantifying Geospatial in the Common Crawl Corpus","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19172","citing_title":"Bridge: Retrieval-Augmented Spatiotemporal Modeling for Urban Delivery Demand","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19343","citing_title":"What Makes a Representation Good for Single-Cell Perturbation Prediction?","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20338","citing_title":"Emergent Manifold Separability during Reasoning in Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":259,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25905","citing_title":"A paradox of AI fluency","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2311.03658","citing_title":"The Linear Representation Hypothesis and the Geometry of Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01609","citing_title":"Concepts Whisper While Syntax Shouts: Spectral Anti-Concentration and the Dual Geometry of Transformer Representations","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19052","citing_title":"Cell-Based Representation of Relational Binding in Language Models","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX","json":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX.json","graph_json":"https://pith.science/api/pith-number/N5HVAVYLAFC56L7NWBRK36RJWX/graph.json","events_json":"https://pith.science/api/pith-number/N5HVAVYLAFC56L7NWBRK36RJWX/events.json","paper":"https://pith.science/paper/N5HVAVYL"},"agent_actions":{"view_html":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX","download_json":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX.json","view_paper":"https://pith.science/paper/N5HVAVYL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.02207&json=true","fetch_graph":"https://pith.science/api/pith-number/N5HVAVYLAFC56L7NWBRK36RJWX/graph.json","fetch_events":"https://pith.science/api/pith-number/N5HVAVYLAFC56L7NWBRK36RJWX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX/action/storage_attestation","attest_author":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX/action/author_attestation","sign_citation":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX/action/citation_signature","submit_replication":"https://pith.science/pith/N5HVAVYLAFC56L7NWBRK36RJWX/action/replication_record"}},"created_at":"2026-07-05T07:51:33.115340+00:00","updated_at":"2026-07-05T07:51:33.115340+00:00"}