{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:P7UCKJ2Q6E3HRDMSS5TOEUROHM","short_pith_number":"pith:P7UCKJ2Q","schema_version":"1.0","canonical_sha256":"7fe8252750f136788d929766e2522e3b10985904437ae9186e4034c150dc2949","source":{"kind":"arxiv","id":"2305.13673","version":4},"attestation_state":"computed","paper":{"title":"Physics of Language Models: Part 1, Learning Hierarchical Language Structures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Yuanzhi Li, Zeyuan Allen-Zhu","submitted_at":"2023-05-23T04:28:16Z","abstract_excerpt":"Transformer-based language models are effective but complex, and understanding their inner workings and reasoning mechanisms is a significant challenge. Previous research has primarily explored how these models handle simple tasks like name copying or selection, and we extend this by investigating how these models perform recursive language structure reasoning defined by context-free grammars (CFGs). We introduce a family of synthetic CFGs that produce hierarchical rules, capable of generating lengthy sentences (e.g., hundreds of tokens) that are locally ambiguous and require dynamic programmi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.13673","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T04:28:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"07a33ee8a037bf589616a0afa0576c41cdfb3ed529bbec4e34f11818c3ab3717","abstract_canon_sha256":"834292fa2ee6db5c86d42d1c52082c4d8892a8e98f185ea71efd8827eac3aeea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:33.841144Z","signature_b64":"DEjsOF9L0LEpV4ag5F8RnMXUKu5d0Frr14CkvgHt9kRlvTIKAmXvdO0KpHNBeHvVgIwDov+XvQbSj6UCCAAZBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fe8252750f136788d929766e2522e3b10985904437ae9186e4034c150dc2949","last_reissued_at":"2026-07-05T11:04:33.840648Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:33.840648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Physics of Language Models: Part 1, Learning Hierarchical Language Structures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Yuanzhi Li, Zeyuan Allen-Zhu","submitted_at":"2023-05-23T04:28:16Z","abstract_excerpt":"Transformer-based language models are effective but complex, and understanding their inner workings and reasoning mechanisms is a significant challenge. Previous research has primarily explored how these models handle simple tasks like name copying or selection, and we extend this by investigating how these models perform recursive language structure reasoning defined by context-free grammars (CFGs). We introduce a family of synthetic CFGs that produce hierarchical rules, capable of generating lengthy sentences (e.g., hundreds of tokens) that are locally ambiguous and require dynamic programmi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.13673","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.13673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.13673","created_at":"2026-07-05T11:04:33.840716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.13673v4","created_at":"2026-07-05T11:04:33.840716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.13673","created_at":"2026-07-05T11:04:33.840716+00:00"},{"alias_kind":"pith_short_12","alias_value":"P7UCKJ2Q6E3H","created_at":"2026-07-05T11:04:33.840716+00:00"},{"alias_kind":"pith_short_16","alias_value":"P7UCKJ2Q6E3HRDMS","created_at":"2026-07-05T11:04:33.840716+00:00"},{"alias_kind":"pith_short_8","alias_value":"P7UCKJ2Q","created_at":"2026-07-05T11:04:33.840716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20347","citing_title":"Critical Percolation as a Synthetic Data Model for Interpretability","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18164","citing_title":"Learning Dynamics of Chain-of-Thought State Tracking in a Solvable Transformer Model","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09822","citing_title":"Causally Evaluating the Learnability of Formal Language Tasks","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08705","citing_title":"Analyzing the Correlation Between Hallucinations and Knowledge Conflicts in Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05025","citing_title":"Invariant Gradient Alignment for Robust Reasoning Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29858","citing_title":"Smooth Scaling Laws Hide Stepwise Token Learning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07568","citing_title":"A Systematic Study of Behavioral Cloning for Scientific Data Annotation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23267","citing_title":"Fine-tuning vs. In-context Learning in Large Language Models: A Formal Language Learning Perspective","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2306.11644","citing_title":"Textbooks Are All You Need","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23267","citing_title":"Fine-tuning vs. In-context Learning in Large Language Models: A Formal Language Learning Perspective","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05495","citing_title":"Shortcut Solutions Learned by Transformers Impair Continual Compositional Reasoning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20811","citing_title":"Diagnosing CFG Interpretation in LLMs","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM","json":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM.json","graph_json":"https://pith.science/api/pith-number/P7UCKJ2Q6E3HRDMSS5TOEUROHM/graph.json","events_json":"https://pith.science/api/pith-number/P7UCKJ2Q6E3HRDMSS5TOEUROHM/events.json","paper":"https://pith.science/paper/P7UCKJ2Q"},"agent_actions":{"view_html":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM","download_json":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM.json","view_paper":"https://pith.science/paper/P7UCKJ2Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.13673&json=true","fetch_graph":"https://pith.science/api/pith-number/P7UCKJ2Q6E3HRDMSS5TOEUROHM/graph.json","fetch_events":"https://pith.science/api/pith-number/P7UCKJ2Q6E3HRDMSS5TOEUROHM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM/action/storage_attestation","attest_author":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM/action/author_attestation","sign_citation":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM/action/citation_signature","submit_replication":"https://pith.science/pith/P7UCKJ2Q6E3HRDMSS5TOEUROHM/action/replication_record"}},"created_at":"2026-07-05T11:04:33.840716+00:00","updated_at":"2026-07-05T11:04:33.840716+00:00"}