{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EF72M3OM5L4T6PX3ELZOBG2AUG","short_pith_number":"pith:EF72M3OM","schema_version":"1.0","canonical_sha256":"217fa66dcceaf93f3efb22f2e09b40a19421d8183fb5b746d16379b8fff3e52f","source":{"kind":"arxiv","id":"2505.04388","version":2},"attestation_state":"computed","paper":{"title":"The Aloe Family Recipe for Open and Specialized Healthcare LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adrian Tormos, Anna Arias-Duart, Ashwin Kumar Gururajan, Daniel Hinjos, Dario Garcia-Gasulla, Eduard Ayguad\\'e-Parra, Enrique Lopez-Cuena, Jordi Bayarri-Planas, Marta Gonzalez-Mallo, Pablo Agustin Martin-Torres, Pablo Bernabeu-Perez, Sergio Alvarez-Napagao, Ulises Cort\\'es","submitted_at":"2025-05-07T13:13:14Z","abstract_excerpt":"Purpose: With advancements in Large Language Models (LLMs) for healthcare, the need arises for competitive open-source models to protect the public interest. This work contributes to the field of open medical LLMs by optimizing key stages of data preprocessing and training, while showing how to improve model safety (through DPO) and efficacy (through RAG). The evaluation methodology used, which includes four different types of tests, defines a new standard for the field. The resultant models, shown to be competitive with the best private alternatives, are released with a permisive license.\n  M"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.04388","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-07T13:13:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a26b69f0fe241d0592db14eb605fc5830dab94b08574f505e928b29d3877eaa8","abstract_canon_sha256":"c4034ac72536641472eba45f25e9d78f15cebad9ae33d5619d68a488f676b240"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:44.793446Z","signature_b64":"LGyw2iDXAA4h3C4Uoina17CNeUPcyCpLyPbtMXnP4t6YVgpDISKOEzClNrSG31sHDQnQ96TmLhN1DOClRPUoBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"217fa66dcceaf93f3efb22f2e09b40a19421d8183fb5b746d16379b8fff3e52f","last_reissued_at":"2026-07-05T11:11:44.792946Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:44.792946Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Aloe Family Recipe for Open and Specialized Healthcare LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adrian Tormos, Anna Arias-Duart, Ashwin Kumar Gururajan, Daniel Hinjos, Dario Garcia-Gasulla, Eduard Ayguad\\'e-Parra, Enrique Lopez-Cuena, Jordi Bayarri-Planas, Marta Gonzalez-Mallo, Pablo Agustin Martin-Torres, Pablo Bernabeu-Perez, Sergio Alvarez-Napagao, Ulises Cort\\'es","submitted_at":"2025-05-07T13:13:14Z","abstract_excerpt":"Purpose: With advancements in Large Language Models (LLMs) for healthcare, the need arises for competitive open-source models to protect the public interest. This work contributes to the field of open medical LLMs by optimizing key stages of data preprocessing and training, while showing how to improve model safety (through DPO) and efficacy (through RAG). The evaluation methodology used, which includes four different types of tests, defines a new standard for the field. The resultant models, shown to be competitive with the best private alternatives, are released with a permisive license.\n  M"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.04388","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.04388/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.04388","created_at":"2026-07-05T11:11:44.793006+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.04388v2","created_at":"2026-07-05T11:11:44.793006+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.04388","created_at":"2026-07-05T11:11:44.793006+00:00"},{"alias_kind":"pith_short_12","alias_value":"EF72M3OM5L4T","created_at":"2026-07-05T11:11:44.793006+00:00"},{"alias_kind":"pith_short_16","alias_value":"EF72M3OM5L4T6PX3","created_at":"2026-07-05T11:11:44.793006+00:00"},{"alias_kind":"pith_short_8","alias_value":"EF72M3OM","created_at":"2026-07-05T11:11:44.793006+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03399","citing_title":"Selective Token-Level Cryptographic Redaction for Privacy-Preserving Clinical Deployment of Large Language Models","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29146","citing_title":"SafeRx-Agent: A Knowledge-Grounded Multi-Agent Framework for Safe and Explainable Medication Recommendation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26577","citing_title":"Benchmarking the Safety of Large Language Models for Robotic Health Attendant Control","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04992","citing_title":"You Snooze, You Lose: Automatic Safety Alignment Restoration through Neural Weight Translation","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG","json":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG.json","graph_json":"https://pith.science/api/pith-number/EF72M3OM5L4T6PX3ELZOBG2AUG/graph.json","events_json":"https://pith.science/api/pith-number/EF72M3OM5L4T6PX3ELZOBG2AUG/events.json","paper":"https://pith.science/paper/EF72M3OM"},"agent_actions":{"view_html":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG","download_json":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG.json","view_paper":"https://pith.science/paper/EF72M3OM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.04388&json=true","fetch_graph":"https://pith.science/api/pith-number/EF72M3OM5L4T6PX3ELZOBG2AUG/graph.json","fetch_events":"https://pith.science/api/pith-number/EF72M3OM5L4T6PX3ELZOBG2AUG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG/action/storage_attestation","attest_author":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG/action/author_attestation","sign_citation":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG/action/citation_signature","submit_replication":"https://pith.science/pith/EF72M3OM5L4T6PX3ELZOBG2AUG/action/replication_record"}},"created_at":"2026-07-05T11:11:44.793006+00:00","updated_at":"2026-07-05T11:11:44.793006+00:00"}