{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OO6EUVYWFKOXHAQYUPJ6UWIWNR","short_pith_number":"pith:OO6EUVYW","schema_version":"1.0","canonical_sha256":"73bc4a57162a9d738218a3d3ea59166c6dc1a7bad0f62f30632b71f9e981ebe1","source":{"kind":"arxiv","id":"2410.00699","version":1},"attestation_state":"computed","paper":{"title":"Investigating the Impact of Model Complexity in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Huiyuan Wang, Jing Luo, Weiran Huang","submitted_at":"2024-10-01T13:53:44Z","abstract_excerpt":"Large Language Models (LLMs) based on the pre-trained fine-tuning paradigm have become pivotal in solving natural language processing tasks, consistently achieving state-of-the-art performance. Nevertheless, the theoretical understanding of how model complexity influences fine-tuning performance remains challenging and has not been well explored yet. In this paper, we focus on autoregressive LLMs and propose to employ Hidden Markov Models (HMMs) to model them. Based on the HMM modeling, we investigate the relationship between model complexity and the generalization capability in downstream tas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.00699","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-01T13:53:44Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"ce0f8d7a094b556087b4edde507ecbafa6a6f3f0708bfa531fbf8eac5466889a","abstract_canon_sha256":"dec756d0ed68b8025e9ed5ec81853ed02015280068cb1a9087ee5e24c411bd0e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:09.175120Z","signature_b64":"soBGCwciezXxndU424/Vv1ZuUHHY1kl+4EbLETtyxOy2AkIfeHaHmA8QQ11xY/BZqDxBmgv3rHxA9Dr4A/xVDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"73bc4a57162a9d738218a3d3ea59166c6dc1a7bad0f62f30632b71f9e981ebe1","last_reissued_at":"2026-07-05T09:14:09.174781Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:09.174781Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating the Impact of Model Complexity in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Huiyuan Wang, Jing Luo, Weiran Huang","submitted_at":"2024-10-01T13:53:44Z","abstract_excerpt":"Large Language Models (LLMs) based on the pre-trained fine-tuning paradigm have become pivotal in solving natural language processing tasks, consistently achieving state-of-the-art performance. Nevertheless, the theoretical understanding of how model complexity influences fine-tuning performance remains challenging and has not been well explored yet. In this paper, we focus on autoregressive LLMs and propose to employ Hidden Markov Models (HMMs) to model them. Based on the HMM modeling, we investigate the relationship between model complexity and the generalization capability in downstream tas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.00699","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.00699/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.00699","created_at":"2026-07-05T09:14:09.174835+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.00699v1","created_at":"2026-07-05T09:14:09.174835+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.00699","created_at":"2026-07-05T09:14:09.174835+00:00"},{"alias_kind":"pith_short_12","alias_value":"OO6EUVYWFKOX","created_at":"2026-07-05T09:14:09.174835+00:00"},{"alias_kind":"pith_short_16","alias_value":"OO6EUVYWFKOXHAQY","created_at":"2026-07-05T09:14:09.174835+00:00"},{"alias_kind":"pith_short_8","alias_value":"OO6EUVYW","created_at":"2026-07-05T09:14:09.174835+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR","json":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR.json","graph_json":"https://pith.science/api/pith-number/OO6EUVYWFKOXHAQYUPJ6UWIWNR/graph.json","events_json":"https://pith.science/api/pith-number/OO6EUVYWFKOXHAQYUPJ6UWIWNR/events.json","paper":"https://pith.science/paper/OO6EUVYW"},"agent_actions":{"view_html":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR","download_json":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR.json","view_paper":"https://pith.science/paper/OO6EUVYW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.00699&json=true","fetch_graph":"https://pith.science/api/pith-number/OO6EUVYWFKOXHAQYUPJ6UWIWNR/graph.json","fetch_events":"https://pith.science/api/pith-number/OO6EUVYWFKOXHAQYUPJ6UWIWNR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR/action/storage_attestation","attest_author":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR/action/author_attestation","sign_citation":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR/action/citation_signature","submit_replication":"https://pith.science/pith/OO6EUVYWFKOXHAQYUPJ6UWIWNR/action/replication_record"}},"created_at":"2026-07-05T09:14:09.174835+00:00","updated_at":"2026-07-05T09:14:09.174835+00:00"}