{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:3OW5NLRGFPKFAZ3Z4RTT7OOFRW","short_pith_number":"pith:3OW5NLRG","schema_version":"1.0","canonical_sha256":"dbadd6ae262bd4506779e4673fb9c58da77c12e1a7eb3134fd78cd4aa56f9558","source":{"kind":"arxiv","id":"2608.10806","version":1},"attestation_state":"computed","paper":{"title":"Assessing Reliability of BERT-Based Models on Question Answering Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Basant Agarwal, Marko Robnik \\v{S}ikonja, Pooja Yadav, Priyanka Harjule","submitted_at":"2026-08-11T11:24:07Z","abstract_excerpt":"Reliability estimation of large language models is in many cases as crucial as their accuracy, as reliable models are more trustworthy, robust, and suitable for practical applications. Recent advancements in natural language processing (NLP), particularly those based on transformer architectures, have significantly accelerated progress across various NLP tasks. This study focuses on the reliability of transformer-based question answering (QA) models, specifically BERT models and its variants (RoBERTa, ALBERT, DistilBERT). These encoder-only pretrained transformers have demonstrated remarkable "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.10806","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-11T11:24:07Z","cross_cats_sorted":[],"title_canon_sha256":"1ed958e2ef4e39c5a83842da2240b3a8228faea111a6f5c307779656dd64af7b","abstract_canon_sha256":"82813566d4b0bc6a1ee566fd26ff1bfa1fdedfa17401e7f4152e3c52c3e5b979"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-12T01:23:19.086624Z","signature_b64":"KrTwCAf0uzMjQzQ4cNvcAPUGZQ0iaDFSJNNt7MUDlv23Rdf0GVhMxkDPvmXA4ihyJ6TloIk78OX/4xG4JSWTCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dbadd6ae262bd4506779e4673fb9c58da77c12e1a7eb3134fd78cd4aa56f9558","last_reissued_at":"2026-08-12T01:23:19.085124Z","signature_status":"signed_v1","first_computed_at":"2026-08-12T01:23:19.085124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assessing Reliability of BERT-Based Models on Question Answering Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Basant Agarwal, Marko Robnik \\v{S}ikonja, Pooja Yadav, Priyanka Harjule","submitted_at":"2026-08-11T11:24:07Z","abstract_excerpt":"Reliability estimation of large language models is in many cases as crucial as their accuracy, as reliable models are more trustworthy, robust, and suitable for practical applications. Recent advancements in natural language processing (NLP), particularly those based on transformer architectures, have significantly accelerated progress across various NLP tasks. This study focuses on the reliability of transformer-based question answering (QA) models, specifically BERT models and its variants (RoBERTa, ALBERT, DistilBERT). These encoder-only pretrained transformers have demonstrated remarkable "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.10806","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.10806/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.10806","created_at":"2026-08-12T01:23:19.088920+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.10806v1","created_at":"2026-08-12T01:23:19.088920+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.10806","created_at":"2026-08-12T01:23:19.088920+00:00"},{"alias_kind":"pith_short_12","alias_value":"3OW5NLRGFPKF","created_at":"2026-08-12T01:23:19.088920+00:00"},{"alias_kind":"pith_short_16","alias_value":"3OW5NLRGFPKFAZ3Z","created_at":"2026-08-12T01:23:19.088920+00:00"},{"alias_kind":"pith_short_8","alias_value":"3OW5NLRG","created_at":"2026-08-12T01:23:19.088920+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW","json":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW.json","graph_json":"https://pith.science/api/pith-number/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/graph.json","events_json":"https://pith.science/api/pith-number/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/events.json","paper":"https://pith.science/paper/3OW5NLRG"},"agent_actions":{"view_html":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW","download_json":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW.json","view_paper":"https://pith.science/paper/3OW5NLRG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.10806&json=true","fetch_graph":"https://pith.science/api/pith-number/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/graph.json","fetch_events":"https://pith.science/api/pith-number/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/action/storage_attestation","attest_author":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/action/author_attestation","sign_citation":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/action/citation_signature","submit_replication":"https://pith.science/pith/3OW5NLRGFPKFAZ3Z4RTT7OOFRW/action/replication_record"}},"created_at":"2026-08-12T01:23:19.088920+00:00","updated_at":"2026-08-12T01:23:19.088920+00:00"}