{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:WUMFYHCKGSXCBUBELJU7XXH2GT","short_pith_number":"pith:WUMFYHCK","schema_version":"1.0","canonical_sha256":"b5185c1c4a34ae20d0245a69fbdcfa34e1248bb0e89ddedabeac137ec2686a22","source":{"kind":"arxiv","id":"2607.00048","version":1},"attestation_state":"computed","paper":{"title":"Comparing Large Language Models on Scrum Certification-Style Questions: Accuracy, Stability, and Error Patterns","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Ademar Fran\\c{c}a de Sousa Neto, Angelo Perkusich, Danyllo Wagner Albuquerque, Emanuel Dantas Filho, Jo\\~ao Paiva, Kyller Gorg\\^onio, Mirko Perkusich, Robson Alves Vilar","submitted_at":"2026-06-29T23:37:56Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used in exam- and certification-style question answering tasks, where their ability to retrieve, interpret, and apply domain-specific knowledge can be systematically assessed. In Software Engineering, such settings are particularly relevant when questions depend on strict adherence to normative definitions, roles, artifacts, and rules. This paper evaluates the performance of three contemporary LLMs, \\textit{GPT-5 mini}, \\textit{Gemini 3 Flash}, and \\textit{DeepSeek Chat 3.2}, in answering 993 Scrum certification-style questions aligned with the Pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.00048","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2026-06-29T23:37:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7533f78d9478327c9196c0cd2a063b17a02d835eb7dc0b5af47fd60b4834aff1","abstract_canon_sha256":"461a02a0611eeb228af4a795614015c50d06e579098fc5557f6fc18147c9137d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-02T00:18:32.482885Z","signature_b64":"x98hSVIG4NIX95ypExaasaWsxJIcQoA3+WBv5ygcLNlJa0i4fI2ucQGTXISbKCTIAKYjxlN/+xAkKQBWmLpCDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b5185c1c4a34ae20d0245a69fbdcfa34e1248bb0e89ddedabeac137ec2686a22","last_reissued_at":"2026-07-02T00:18:32.482330Z","signature_status":"signed_v1","first_computed_at":"2026-07-02T00:18:32.482330Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Comparing Large Language Models on Scrum Certification-Style Questions: Accuracy, Stability, and Error Patterns","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Ademar Fran\\c{c}a de Sousa Neto, Angelo Perkusich, Danyllo Wagner Albuquerque, Emanuel Dantas Filho, Jo\\~ao Paiva, Kyller Gorg\\^onio, Mirko Perkusich, Robson Alves Vilar","submitted_at":"2026-06-29T23:37:56Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used in exam- and certification-style question answering tasks, where their ability to retrieve, interpret, and apply domain-specific knowledge can be systematically assessed. In Software Engineering, such settings are particularly relevant when questions depend on strict adherence to normative definitions, roles, artifacts, and rules. This paper evaluates the performance of three contemporary LLMs, \\textit{GPT-5 mini}, \\textit{Gemini 3 Flash}, and \\textit{DeepSeek Chat 3.2}, in answering 993 Scrum certification-style questions aligned with the Pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.00048","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.00048/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.00048","created_at":"2026-07-02T00:18:32.482428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.00048v1","created_at":"2026-07-02T00:18:32.482428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.00048","created_at":"2026-07-02T00:18:32.482428+00:00"},{"alias_kind":"pith_short_12","alias_value":"WUMFYHCKGSXC","created_at":"2026-07-02T00:18:32.482428+00:00"},{"alias_kind":"pith_short_16","alias_value":"WUMFYHCKGSXCBUBE","created_at":"2026-07-02T00:18:32.482428+00:00"},{"alias_kind":"pith_short_8","alias_value":"WUMFYHCK","created_at":"2026-07-02T00:18:32.482428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT","json":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT.json","graph_json":"https://pith.science/api/pith-number/WUMFYHCKGSXCBUBELJU7XXH2GT/graph.json","events_json":"https://pith.science/api/pith-number/WUMFYHCKGSXCBUBELJU7XXH2GT/events.json","paper":"https://pith.science/paper/WUMFYHCK"},"agent_actions":{"view_html":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT","download_json":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT.json","view_paper":"https://pith.science/paper/WUMFYHCK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.00048&json=true","fetch_graph":"https://pith.science/api/pith-number/WUMFYHCKGSXCBUBELJU7XXH2GT/graph.json","fetch_events":"https://pith.science/api/pith-number/WUMFYHCKGSXCBUBELJU7XXH2GT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT/action/storage_attestation","attest_author":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT/action/author_attestation","sign_citation":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT/action/citation_signature","submit_replication":"https://pith.science/pith/WUMFYHCKGSXCBUBELJU7XXH2GT/action/replication_record"}},"created_at":"2026-07-02T00:18:32.482428+00:00","updated_at":"2026-07-02T00:18:32.482428+00:00"}