{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PYUDPFWAGVYTUDZHRFMDE5IXMF","short_pith_number":"pith:PYUDPFWA","schema_version":"1.0","canonical_sha256":"7e283796c035713a0f27895832751761766871530fec74dd0cb128c882230156","source":{"kind":"arxiv","id":"2510.18315","version":2},"attestation_state":"computed","paper":{"title":"Higher Embedding Dimension Creates a Stronger World Model for a Simple Sorting Task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Brady Bhalla, Honglu Fan, Nancy Chen, Tony Yue Yu","submitted_at":"2025-10-21T05:51:02Z","abstract_excerpt":"We investigate how embedding dimension affects the emergence of an internal \"world model\" in a transformer trained with reinforcement learning to perform bubble-sort-style adjacent swaps. Models achieve high accuracy even with very small embedding dimensions, but larger dimensions yield more faithful, consistent, and robust internal representations. In particular, higher embedding dimensions strengthen the formation of structured internal representation and lead to better interpretability. After hundreds of experiments, we observe two consistent mechanisms: (1) the last row of the attention we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.18315","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-10-21T05:51:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1e2328db810fb96bdc9c5795eb74a56c1e616be4a4b86050e1ec9838ca183b3c","abstract_canon_sha256":"7af7e85c7ddff8e1f3e2c73bdaa202d66cad264a3c6562b11fe29864b6c3e758"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T01:20:48.933059Z","signature_b64":"X8evCrprTic9HaPUE8Syw9ueBXx6JXT+/3BqZsf/nBcSauokEf6kNXCmbpcBFve/HAOBJRxR3K6rfA0iVnluDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e283796c035713a0f27895832751761766871530fec74dd0cb128c882230156","last_reissued_at":"2026-07-15T01:20:48.932186Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T01:20:48.932186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Higher Embedding Dimension Creates a Stronger World Model for a Simple Sorting Task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Brady Bhalla, Honglu Fan, Nancy Chen, Tony Yue Yu","submitted_at":"2025-10-21T05:51:02Z","abstract_excerpt":"We investigate how embedding dimension affects the emergence of an internal \"world model\" in a transformer trained with reinforcement learning to perform bubble-sort-style adjacent swaps. Models achieve high accuracy even with very small embedding dimensions, but larger dimensions yield more faithful, consistent, and robust internal representations. In particular, higher embedding dimensions strengthen the formation of structured internal representation and lead to better interpretability. After hundreds of experiments, we observe two consistent mechanisms: (1) the last row of the attention we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.18315","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.18315/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.18315","created_at":"2026-07-15T01:20:48.932604+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.18315v2","created_at":"2026-07-15T01:20:48.932604+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.18315","created_at":"2026-07-15T01:20:48.932604+00:00"},{"alias_kind":"pith_short_12","alias_value":"PYUDPFWAGVYT","created_at":"2026-07-15T01:20:48.932604+00:00"},{"alias_kind":"pith_short_16","alias_value":"PYUDPFWAGVYTUDZH","created_at":"2026-07-15T01:20:48.932604+00:00"},{"alias_kind":"pith_short_8","alias_value":"PYUDPFWA","created_at":"2026-07-15T01:20:48.932604+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.08950","citing_title":"When More Cores Hurts: The Vector Database Scaling Paradox in HPC","ref_index":112,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF","json":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF.json","graph_json":"https://pith.science/api/pith-number/PYUDPFWAGVYTUDZHRFMDE5IXMF/graph.json","events_json":"https://pith.science/api/pith-number/PYUDPFWAGVYTUDZHRFMDE5IXMF/events.json","paper":"https://pith.science/paper/PYUDPFWA"},"agent_actions":{"view_html":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF","download_json":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF.json","view_paper":"https://pith.science/paper/PYUDPFWA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.18315&json=true","fetch_graph":"https://pith.science/api/pith-number/PYUDPFWAGVYTUDZHRFMDE5IXMF/graph.json","fetch_events":"https://pith.science/api/pith-number/PYUDPFWAGVYTUDZHRFMDE5IXMF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF/action/storage_attestation","attest_author":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF/action/author_attestation","sign_citation":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF/action/citation_signature","submit_replication":"https://pith.science/pith/PYUDPFWAGVYTUDZHRFMDE5IXMF/action/replication_record"}},"created_at":"2026-07-15T01:20:48.932604+00:00","updated_at":"2026-07-15T01:20:48.932604+00:00"}