{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:I7FIBYSKD26NHSCXTLHA74Y6VR","short_pith_number":"pith:I7FIBYSK","schema_version":"1.0","canonical_sha256":"47ca80e24a1ebcd3c8579ace0ff31eac4ac416b12bc487148f695b937892ea12","source":{"kind":"arxiv","id":"2307.09793","version":1},"attestation_state":"computed","paper":{"title":"On the Origin of LLMs: An Evolutionary Tree and Graph for 15,821 Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.DL","authors_text":"Andrew Kean Gao, Sarah Gao","submitted_at":"2023-07-19T07:17:43Z","abstract_excerpt":"Since late 2022, Large Language Models (LLMs) have become very prominent with LLMs like ChatGPT and Bard receiving millions of users. Hundreds of new LLMs are announced each week, many of which are deposited to Hugging Face, a repository of machine learning models and datasets. To date, nearly 16,000 Text Generation models have been uploaded to the site. Given the huge influx of LLMs, it is of interest to know which LLM backbones, settings, training methods, and families are popular or trending. However, there is no comprehensive index of LLMs available. We take advantage of the relatively sys"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.09793","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.DL","submitted_at":"2023-07-19T07:17:43Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"044e8c472d24273e97003cff37cefc95d227a2127051806aca72383fc4348e06","abstract_canon_sha256":"b54c8ca71cc46a2e97821f4603486ebc893cbf6a4930a7a9c874124c5f52680f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:32:46.388409Z","signature_b64":"Cd/7d7+hPMaf82FTKf4W/txL5SeOw9qdMS2uAcnSCkPFCgUvlbb98ou4phtQs3dlmpdNO4s9SreW+qiJRmbLCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47ca80e24a1ebcd3c8579ace0ff31eac4ac416b12bc487148f695b937892ea12","last_reissued_at":"2026-07-05T06:32:46.387970Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:32:46.387970Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Origin of LLMs: An Evolutionary Tree and Graph for 15,821 Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.DL","authors_text":"Andrew Kean Gao, Sarah Gao","submitted_at":"2023-07-19T07:17:43Z","abstract_excerpt":"Since late 2022, Large Language Models (LLMs) have become very prominent with LLMs like ChatGPT and Bard receiving millions of users. Hundreds of new LLMs are announced each week, many of which are deposited to Hugging Face, a repository of machine learning models and datasets. To date, nearly 16,000 Text Generation models have been uploaded to the site. Given the huge influx of LLMs, it is of interest to know which LLM backbones, settings, training methods, and families are popular or trending. However, there is no comprehensive index of LLMs available. We take advantage of the relatively sys"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.09793","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.09793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.09793","created_at":"2026-07-05T06:32:46.388026+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.09793v1","created_at":"2026-07-05T06:32:46.388026+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.09793","created_at":"2026-07-05T06:32:46.388026+00:00"},{"alias_kind":"pith_short_12","alias_value":"I7FIBYSKD26N","created_at":"2026-07-05T06:32:46.388026+00:00"},{"alias_kind":"pith_short_16","alias_value":"I7FIBYSKD26NHSCX","created_at":"2026-07-05T06:32:46.388026+00:00"},{"alias_kind":"pith_short_8","alias_value":"I7FIBYSK","created_at":"2026-07-05T06:32:46.388026+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16902","citing_title":"ArtifactLinker: Linking Scientific Artifacts for Automatic State-of-the-Art Discovery","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR","json":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR.json","graph_json":"https://pith.science/api/pith-number/I7FIBYSKD26NHSCXTLHA74Y6VR/graph.json","events_json":"https://pith.science/api/pith-number/I7FIBYSKD26NHSCXTLHA74Y6VR/events.json","paper":"https://pith.science/paper/I7FIBYSK"},"agent_actions":{"view_html":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR","download_json":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR.json","view_paper":"https://pith.science/paper/I7FIBYSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.09793&json=true","fetch_graph":"https://pith.science/api/pith-number/I7FIBYSKD26NHSCXTLHA74Y6VR/graph.json","fetch_events":"https://pith.science/api/pith-number/I7FIBYSKD26NHSCXTLHA74Y6VR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR/action/storage_attestation","attest_author":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR/action/author_attestation","sign_citation":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR/action/citation_signature","submit_replication":"https://pith.science/pith/I7FIBYSKD26NHSCXTLHA74Y6VR/action/replication_record"}},"created_at":"2026-07-05T06:32:46.388026+00:00","updated_at":"2026-07-05T06:32:46.388026+00:00"}