{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UOFSBVRJ6GHILECMRKS2DOZPPF","short_pith_number":"pith:UOFSBVRJ","schema_version":"1.0","canonical_sha256":"a38b20d629f18e85904c8aa5a1bb2f794473eec42b9ec722abf163f214523ebc","source":{"kind":"arxiv","id":"2305.18855","version":1},"attestation_state":"computed","paper":{"title":"STT4SG-350: A Speech Corpus for All Swiss German Dialect Regions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Christian Scheller, Claudio Paonessa, Jan Deriu, Julia Hartmann, Larissa Schmidt, Manfred Vogel, Manuela H\\\"urlimann, Mark Cieliebak, Michel Pl\\\"uss, Tanja Samard\\v{z}i\\'c, Yanick Schraner","submitted_at":"2023-05-30T08:49:38Z","abstract_excerpt":"We present STT4SG-350 (Speech-to-Text for Swiss German), a corpus of Swiss German speech, annotated with Standard German text at the sentence level. The data is collected using a web app in which the speakers are shown Standard German sentences, which they translate to Swiss German and record. We make the corpus publicly available. It contains 343 hours of speech from all dialect regions and is the largest public speech corpus for Swiss German to date. Application areas include automatic speech recognition (ASR), text-to-speech, dialect identification, and speaker recognition. Dialect informat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18855","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-30T08:49:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1df2b75ed991b537a1c092f6d7c91679c9492e72e21c08c925e261c7dedff85b","abstract_canon_sha256":"84f15488c8e2a55acdc851658eec19af64aeee6a69b90fb4810676843454f0e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:33.987482Z","signature_b64":"Z68PZ9o0ujixkqsGeqTsHUfL4a0yAyrM/dTwNA2YKExK4mp4qkgXuIu0F7Upgf3+2uc0eW5ViDY26DunXStkCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a38b20d629f18e85904c8aa5a1bb2f794473eec42b9ec722abf163f214523ebc","last_reissued_at":"2026-07-05T06:15:33.987014Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:33.987014Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"STT4SG-350: A Speech Corpus for All Swiss German Dialect Regions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Christian Scheller, Claudio Paonessa, Jan Deriu, Julia Hartmann, Larissa Schmidt, Manfred Vogel, Manuela H\\\"urlimann, Mark Cieliebak, Michel Pl\\\"uss, Tanja Samard\\v{z}i\\'c, Yanick Schraner","submitted_at":"2023-05-30T08:49:38Z","abstract_excerpt":"We present STT4SG-350 (Speech-to-Text for Swiss German), a corpus of Swiss German speech, annotated with Standard German text at the sentence level. The data is collected using a web app in which the speakers are shown Standard German sentences, which they translate to Swiss German and record. We make the corpus publicly available. It contains 343 hours of speech from all dialect regions and is the largest public speech corpus for Swiss German to date. Application areas include automatic speech recognition (ASR), text-to-speech, dialect identification, and speaker recognition. Dialect informat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18855","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18855/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18855","created_at":"2026-07-05T06:15:33.987067+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18855v1","created_at":"2026-07-05T06:15:33.987067+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18855","created_at":"2026-07-05T06:15:33.987067+00:00"},{"alias_kind":"pith_short_12","alias_value":"UOFSBVRJ6GHI","created_at":"2026-07-05T06:15:33.987067+00:00"},{"alias_kind":"pith_short_16","alias_value":"UOFSBVRJ6GHILECM","created_at":"2026-07-05T06:15:33.987067+00:00"},{"alias_kind":"pith_short_8","alias_value":"UOFSBVRJ","created_at":"2026-07-05T06:15:33.987067+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF","json":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF.json","graph_json":"https://pith.science/api/pith-number/UOFSBVRJ6GHILECMRKS2DOZPPF/graph.json","events_json":"https://pith.science/api/pith-number/UOFSBVRJ6GHILECMRKS2DOZPPF/events.json","paper":"https://pith.science/paper/UOFSBVRJ"},"agent_actions":{"view_html":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF","download_json":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF.json","view_paper":"https://pith.science/paper/UOFSBVRJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18855&json=true","fetch_graph":"https://pith.science/api/pith-number/UOFSBVRJ6GHILECMRKS2DOZPPF/graph.json","fetch_events":"https://pith.science/api/pith-number/UOFSBVRJ6GHILECMRKS2DOZPPF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF/action/storage_attestation","attest_author":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF/action/author_attestation","sign_citation":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF/action/citation_signature","submit_replication":"https://pith.science/pith/UOFSBVRJ6GHILECMRKS2DOZPPF/action/replication_record"}},"created_at":"2026-07-05T06:15:33.987067+00:00","updated_at":"2026-07-05T06:15:33.987067+00:00"}