{"work":{"id":"07bf7000-64b4-4c9d-9f3d-cac739d79cf3","openalex_id":"https://openalex.org/W4391514311","doi":"10.48550/arxiv.2402.00159","arxiv_id":"2402.00159","raw_key":null,"title":"Dolma: An open cor- pus of three trillion tokens for language model pretraining research, 2024","authors":null,"authors_text":"Luca Soldaini, Rodney Kinney, Akshita Bhagia, Dustin Schwenk, David Atkinson, Russell Authur, Ben Bogin, Khyathi Chandu, Jennifer Dumas, Yanai Elazar, Valentin Hofmann, Ananya Harsh Jha, Sachin Kumar, Lucy Li, Xinxi Lyu, Nathan Lambert, Ian","year":2024,"venue":"arXiv (Cornell University)","abstract":null,"external_url":"https://arxiv.org/abs/2402.00159","cited_by_count":9,"metadata_source":"arxiv_reference","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":null,"created_at":"2026-05-09T20:37:32.061820+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Dolma: An open corpus of three trillion tokens for language model pretraining research","render_title":"Dolma: An open corpus of three trillion tokens for language model pretraining research"},"hub":{"state":{"work_id":"07bf7000-64b4-4c9d-9f3d-cac739d79cf3","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":23,"external_cited_by_count":9,"distinct_field_count":5,"first_pith_cited_at":"2023-03-31T17:28:46+00:00","last_pith_cited_at":"2026-06-30T23:52:15+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T19:19:48.211015+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":1},{"context_role":"dataset","n":1}],"polarity_counts":[{"context_polarity":"background","n":1},{"context_polarity":"use_dataset","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}