{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VIDALH2ZPQGBCMTAKD26GKQBCR","short_pith_number":"pith:VIDALH2Z","schema_version":"1.0","canonical_sha256":"aa06059f597c0c11326050f5e32a0114512938b5fa93635198649bdda4e5a05d","source":{"kind":"arxiv","id":"2506.20331","version":1},"attestation_state":"computed","paper":{"title":"Biomed-Enriched: A Biomedical Dataset Enriched with LLMs for Pretraining and Extracting Rare and Hidden Content","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Eric de la Clergerie, Nathan Godey, Rian Touchent","submitted_at":"2025-06-25T11:30:25Z","abstract_excerpt":"We introduce Biomed-Enriched, a biomedical text dataset constructed from PubMed via a two-stage annotation process. In the first stage, a large language model annotates 400K paragraphs from PubMed scientific articles, assigning scores for their type (review, study, clinical case, other), domain (clinical, biomedical, other), and educational quality. The educational quality score (rated 1 to 5) estimates how useful a paragraph is for college-level learning. These annotations are then used to fine-tune a small language model, which propagates the labels across the full PMC-OA corpus. The resulti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.20331","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-25T11:30:25Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"048b3987078ff875666551a5966ac882dc8db092c8d9614241dfa3e716262c94","abstract_canon_sha256":"06ddb4b23b88b0a819aed9705bbd255e54c67aa14a27f830901d82c3bcaa51b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:03.880383Z","signature_b64":"O2SB/oDOUSGFrb3Rxtxwj7Shrprpu1POTRuSH0lcRfA6J8LMPpfkHUeFMyJEzP8CVsjXCARuEOyxa1noIq82AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa06059f597c0c11326050f5e32a0114512938b5fa93635198649bdda4e5a05d","last_reissued_at":"2026-07-05T11:27:03.879979Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:03.879979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Biomed-Enriched: A Biomedical Dataset Enriched with LLMs for Pretraining and Extracting Rare and Hidden Content","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Eric de la Clergerie, Nathan Godey, Rian Touchent","submitted_at":"2025-06-25T11:30:25Z","abstract_excerpt":"We introduce Biomed-Enriched, a biomedical text dataset constructed from PubMed via a two-stage annotation process. In the first stage, a large language model annotates 400K paragraphs from PubMed scientific articles, assigning scores for their type (review, study, clinical case, other), domain (clinical, biomedical, other), and educational quality. The educational quality score (rated 1 to 5) estimates how useful a paragraph is for college-level learning. These annotations are then used to fine-tune a small language model, which propagates the labels across the full PMC-OA corpus. The resulti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.20331","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.20331/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.20331","created_at":"2026-07-05T11:27:03.880036+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.20331v1","created_at":"2026-07-05T11:27:03.880036+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.20331","created_at":"2026-07-05T11:27:03.880036+00:00"},{"alias_kind":"pith_short_12","alias_value":"VIDALH2ZPQGB","created_at":"2026-07-05T11:27:03.880036+00:00"},{"alias_kind":"pith_short_16","alias_value":"VIDALH2ZPQGBCMTA","created_at":"2026-07-05T11:27:03.880036+00:00"},{"alias_kind":"pith_short_8","alias_value":"VIDALH2Z","created_at":"2026-07-05T11:27:03.880036+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12438","citing_title":"A Causal Language Modeling Detour Improves Encoder Continued Pretraining","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR","json":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR.json","graph_json":"https://pith.science/api/pith-number/VIDALH2ZPQGBCMTAKD26GKQBCR/graph.json","events_json":"https://pith.science/api/pith-number/VIDALH2ZPQGBCMTAKD26GKQBCR/events.json","paper":"https://pith.science/paper/VIDALH2Z"},"agent_actions":{"view_html":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR","download_json":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR.json","view_paper":"https://pith.science/paper/VIDALH2Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.20331&json=true","fetch_graph":"https://pith.science/api/pith-number/VIDALH2ZPQGBCMTAKD26GKQBCR/graph.json","fetch_events":"https://pith.science/api/pith-number/VIDALH2ZPQGBCMTAKD26GKQBCR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR/action/storage_attestation","attest_author":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR/action/author_attestation","sign_citation":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR/action/citation_signature","submit_replication":"https://pith.science/pith/VIDALH2ZPQGBCMTAKD26GKQBCR/action/replication_record"}},"created_at":"2026-07-05T11:27:03.880036+00:00","updated_at":"2026-07-05T11:27:03.880036+00:00"}