{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WWB52JROXERFDXBAMIWOVRJZ3G","short_pith_number":"pith:WWB52JRO","schema_version":"1.0","canonical_sha256":"b583dd262eb92251dc20622ceac539d98c9098c86ee405a8743ab42e359f1c67","source":{"kind":"arxiv","id":"2505.22759","version":2},"attestation_state":"computed","paper":{"title":"FAMA: The First Large-Scale Open-Science Speech Foundation Model for English and Italian","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"cs.CL","authors_text":"Alessio Brutti, Luisa Bentivogli, Marco Gaido, Marco Matassoni, Matteo Negri, Mauro Cettolo, Mohamed Nabih, Roberto Gretter, Sara Papi","submitted_at":"2025-05-28T18:19:34Z","abstract_excerpt":"The development of speech foundation models (SFMs) like Whisper and SeamlessM4T has significantly advanced the field of speech processing. However, their closed nature--with inaccessible training data and code--poses major reproducibility and fair evaluation challenges. While other domains have made substantial progress toward open science by developing fully transparent models trained on open-source (OS) code and data, similar efforts in speech remain limited. To fill this gap, we introduce FAMA, the first family of open science SFMs for English and Italian, trained on 150k+ hours of OS speec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22759","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T18:19:34Z","cross_cats_sorted":["cs.AI","cs.SD"],"title_canon_sha256":"025491dd493b68fc440ae32f731637ad97d8b7b85d5f15795088bd6b9cf6a16c","abstract_canon_sha256":"0b39cc4af7f499aa9955a3b42cfff3afbb1be5f2f6487771773fb5f38d655a6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:27.604232Z","signature_b64":"U9P06L1xllHJBHhB5Xvzdrjnb1XK4K8y3kdVdIyUsrs8l6SGArp+pO5tVR9dOfu4tLaToeiayBCvN50Q6NSICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b583dd262eb92251dc20622ceac539d98c9098c86ee405a8743ab42e359f1c67","last_reissued_at":"2026-07-05T11:13:27.603692Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:27.603692Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FAMA: The First Large-Scale Open-Science Speech Foundation Model for English and Italian","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"cs.CL","authors_text":"Alessio Brutti, Luisa Bentivogli, Marco Gaido, Marco Matassoni, Matteo Negri, Mauro Cettolo, Mohamed Nabih, Roberto Gretter, Sara Papi","submitted_at":"2025-05-28T18:19:34Z","abstract_excerpt":"The development of speech foundation models (SFMs) like Whisper and SeamlessM4T has significantly advanced the field of speech processing. However, their closed nature--with inaccessible training data and code--poses major reproducibility and fair evaluation challenges. While other domains have made substantial progress toward open science by developing fully transparent models trained on open-source (OS) code and data, similar efforts in speech remain limited. To fill this gap, we introduce FAMA, the first family of open science SFMs for English and Italian, trained on 150k+ hours of OS speec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22759","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22759/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22759","created_at":"2026-07-05T11:13:27.603762+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22759v2","created_at":"2026-07-05T11:13:27.603762+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22759","created_at":"2026-07-05T11:13:27.603762+00:00"},{"alias_kind":"pith_short_12","alias_value":"WWB52JROXERF","created_at":"2026-07-05T11:13:27.603762+00:00"},{"alias_kind":"pith_short_16","alias_value":"WWB52JROXERFDXBA","created_at":"2026-07-05T11:13:27.603762+00:00"},{"alias_kind":"pith_short_8","alias_value":"WWB52JRO","created_at":"2026-07-05T11:13:27.603762+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G","json":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G.json","graph_json":"https://pith.science/api/pith-number/WWB52JROXERFDXBAMIWOVRJZ3G/graph.json","events_json":"https://pith.science/api/pith-number/WWB52JROXERFDXBAMIWOVRJZ3G/events.json","paper":"https://pith.science/paper/WWB52JRO"},"agent_actions":{"view_html":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G","download_json":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G.json","view_paper":"https://pith.science/paper/WWB52JRO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22759&json=true","fetch_graph":"https://pith.science/api/pith-number/WWB52JROXERFDXBAMIWOVRJZ3G/graph.json","fetch_events":"https://pith.science/api/pith-number/WWB52JROXERFDXBAMIWOVRJZ3G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G/action/storage_attestation","attest_author":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G/action/author_attestation","sign_citation":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G/action/citation_signature","submit_replication":"https://pith.science/pith/WWB52JROXERFDXBAMIWOVRJZ3G/action/replication_record"}},"created_at":"2026-07-05T11:13:27.603762+00:00","updated_at":"2026-07-05T11:13:27.603762+00:00"}