{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:IW4HMRDFMZMZXIWZK353NZG6O2","short_pith_number":"pith:IW4HMRDF","schema_version":"1.0","canonical_sha256":"45b876446566599ba2d956fbb6e4de7695b594fe106e2b47991e06d73f6aa030","source":{"kind":"arxiv","id":"2201.10881","version":1},"attestation_state":"computed","paper":{"title":"The Norwegian Parliamentary Speech Corpus","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Pablo Ortiz, Per Erik Solberg","submitted_at":"2022-01-26T11:41:55Z","abstract_excerpt":"The Norwegian Parliamentary Speech Corpus (NPSC) is a speech dataset with recordings of meetings from Stortinget, the Norwegian parliament. It is the first, publicly available dataset containing unscripted, Norwegian speech designed for training of automatic speech recognition (ASR) systems. The recordings are manually transcribed and annotated with language codes and speakers, and there are detailed metadata about the speakers. The transcriptions exist in both normalized and non-normalized form, and non-standardized words are explicitly marked and annotated with standardized equivalents. To t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.10881","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-01-26T11:41:55Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"2181183b4e5ff1ecba01c2ba1ffd9596753af5c9f3273f50e6f12f00f329ca18","abstract_canon_sha256":"29ad3a77230cf18fe37273025696c1f208b530d084a5ffac6b33e4c4a27ca8e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:39:11.994983Z","signature_b64":"cUkSxJp0+WJQmFHD2inRXVxvak1lN7YUSTDGQBkcWhNNLIntjrnzlQOpqGMmO2otuknfbmxF8paMTRtH3Kh/Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45b876446566599ba2d956fbb6e4de7695b594fe106e2b47991e06d73f6aa030","last_reissued_at":"2026-07-05T05:39:11.994515Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:39:11.994515Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Norwegian Parliamentary Speech Corpus","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Pablo Ortiz, Per Erik Solberg","submitted_at":"2022-01-26T11:41:55Z","abstract_excerpt":"The Norwegian Parliamentary Speech Corpus (NPSC) is a speech dataset with recordings of meetings from Stortinget, the Norwegian parliament. It is the first, publicly available dataset containing unscripted, Norwegian speech designed for training of automatic speech recognition (ASR) systems. The recordings are manually transcribed and annotated with language codes and speakers, and there are detailed metadata about the speakers. The transcriptions exist in both normalized and non-normalized form, and non-standardized words are explicitly marked and annotated with standardized equivalents. To t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.10881","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.10881/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.10881","created_at":"2026-07-05T05:39:11.994565+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.10881v1","created_at":"2026-07-05T05:39:11.994565+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.10881","created_at":"2026-07-05T05:39:11.994565+00:00"},{"alias_kind":"pith_short_12","alias_value":"IW4HMRDFMZMZ","created_at":"2026-07-05T05:39:11.994565+00:00"},{"alias_kind":"pith_short_16","alias_value":"IW4HMRDFMZMZXIWZ","created_at":"2026-07-05T05:39:11.994565+00:00"},{"alias_kind":"pith_short_8","alias_value":"IW4HMRDF","created_at":"2026-07-05T05:39:11.994565+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.01439","citing_title":"Whale: Large-Scale multilingual ASR model with w2v-BERT and E-Branchformer with large speech data","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2","json":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2.json","graph_json":"https://pith.science/api/pith-number/IW4HMRDFMZMZXIWZK353NZG6O2/graph.json","events_json":"https://pith.science/api/pith-number/IW4HMRDFMZMZXIWZK353NZG6O2/events.json","paper":"https://pith.science/paper/IW4HMRDF"},"agent_actions":{"view_html":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2","download_json":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2.json","view_paper":"https://pith.science/paper/IW4HMRDF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.10881&json=true","fetch_graph":"https://pith.science/api/pith-number/IW4HMRDFMZMZXIWZK353NZG6O2/graph.json","fetch_events":"https://pith.science/api/pith-number/IW4HMRDFMZMZXIWZK353NZG6O2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2/action/storage_attestation","attest_author":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2/action/author_attestation","sign_citation":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2/action/citation_signature","submit_replication":"https://pith.science/pith/IW4HMRDFMZMZXIWZK353NZG6O2/action/replication_record"}},"created_at":"2026-07-05T05:39:11.994565+00:00","updated_at":"2026-07-05T05:39:11.994565+00:00"}