{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TA5G6RKKTABEVB5TQS3LUYQFEI","short_pith_number":"pith:TA5G6RKK","schema_version":"1.0","canonical_sha256":"983a6f454a98024a87b384b6ba62052217829d72953a816b92f56878a4d6598a","source":{"kind":"arxiv","id":"2303.00293","version":1},"attestation_state":"computed","paper":{"title":"How Robust is GPT-3.5 to Predecessors? A Comprehensive Study on Language Understanding Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Can Zu, Jie Zhou, Junjie Ye, Minlong Peng, Nuo Xu, Qi Zhang, Rui Zheng, Tao Gui, Xuanjing Huang, Xuanting Chen","submitted_at":"2023-03-01T07:39:01Z","abstract_excerpt":"The GPT-3.5 models have demonstrated impressive performance in various Natural Language Processing (NLP) tasks, showcasing their strong understanding and reasoning capabilities. However, their robustness and abilities to handle various complexities of the open world have yet to be explored, which is especially crucial in assessing the stability of models and is a key aspect of trustworthy AI. In this study, we perform a comprehensive experimental analysis of GPT-3.5, exploring its robustness using 21 datasets (about 116K test samples) with 66 text transformations from TextFlint that cover 9 po"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.00293","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-03-01T07:39:01Z","cross_cats_sorted":[],"title_canon_sha256":"72a9445cf2de1ce1f79958fac8b854c7557085c81a260a418453831b5d6d8e43","abstract_canon_sha256":"74176fad09d4b061da29f7c1441003da5743312eabb2b9f4ebf209c9b73cdd51"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:47:07.651494Z","signature_b64":"2FB88vHTroDlH8Ly3tkJUWfWRusF8wEzFSk/qzG5RHmvIbMc2ehhyCXXWEqg1Ab7RPCkieY4v2oMyIvNGgrkDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"983a6f454a98024a87b384b6ba62052217829d72953a816b92f56878a4d6598a","last_reissued_at":"2026-07-05T05:47:07.651111Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:47:07.651111Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Robust is GPT-3.5 to Predecessors? A Comprehensive Study on Language Understanding Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Can Zu, Jie Zhou, Junjie Ye, Minlong Peng, Nuo Xu, Qi Zhang, Rui Zheng, Tao Gui, Xuanjing Huang, Xuanting Chen","submitted_at":"2023-03-01T07:39:01Z","abstract_excerpt":"The GPT-3.5 models have demonstrated impressive performance in various Natural Language Processing (NLP) tasks, showcasing their strong understanding and reasoning capabilities. However, their robustness and abilities to handle various complexities of the open world have yet to be explored, which is especially crucial in assessing the stability of models and is a key aspect of trustworthy AI. In this study, we perform a comprehensive experimental analysis of GPT-3.5, exploring its robustness using 21 datasets (about 116K test samples) with 66 text transformations from TextFlint that cover 9 po"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.00293","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.00293/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.00293","created_at":"2026-07-05T05:47:07.651170+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.00293v1","created_at":"2026-07-05T05:47:07.651170+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.00293","created_at":"2026-07-05T05:47:07.651170+00:00"},{"alias_kind":"pith_short_12","alias_value":"TA5G6RKKTABE","created_at":"2026-07-05T05:47:07.651170+00:00"},{"alias_kind":"pith_short_16","alias_value":"TA5G6RKKTABEVB5T","created_at":"2026-07-05T05:47:07.651170+00:00"},{"alias_kind":"pith_short_8","alias_value":"TA5G6RKK","created_at":"2026-07-05T05:47:07.651170+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.04665","citing_title":"Paraphrase-Induced Output-Mode Collapse: When LLMs Break Character Under Semantically Equivalent Inputs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04665","citing_title":"Paraphrase-Induced Output-Mode Collapse: When LLMs Break Character Under Semantically Equivalent Inputs","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI","json":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI.json","graph_json":"https://pith.science/api/pith-number/TA5G6RKKTABEVB5TQS3LUYQFEI/graph.json","events_json":"https://pith.science/api/pith-number/TA5G6RKKTABEVB5TQS3LUYQFEI/events.json","paper":"https://pith.science/paper/TA5G6RKK"},"agent_actions":{"view_html":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI","download_json":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI.json","view_paper":"https://pith.science/paper/TA5G6RKK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.00293&json=true","fetch_graph":"https://pith.science/api/pith-number/TA5G6RKKTABEVB5TQS3LUYQFEI/graph.json","fetch_events":"https://pith.science/api/pith-number/TA5G6RKKTABEVB5TQS3LUYQFEI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI/action/storage_attestation","attest_author":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI/action/author_attestation","sign_citation":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI/action/citation_signature","submit_replication":"https://pith.science/pith/TA5G6RKKTABEVB5TQS3LUYQFEI/action/replication_record"}},"created_at":"2026-07-05T05:47:07.651170+00:00","updated_at":"2026-07-05T05:47:07.651170+00:00"}