{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5LQ7GNWEBKOUEAKJBU2S2MU5TO","short_pith_number":"pith:5LQ7GNWE","schema_version":"1.0","canonical_sha256":"eae1f336c40a9d4201490d352d329d9bb3a70cc96e5588b2482993e7b6345775","source":{"kind":"arxiv","id":"2307.09009","version":3},"attestation_state":"computed","paper":{"title":"How is ChatGPT's behavior changing over time?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"James Zou, Lingjiao Chen, Matei Zaharia","submitted_at":"2023-07-18T06:56:08Z","abstract_excerpt":"GPT-3.5 and GPT-4 are the two most widely used large language model (LLM) services. However, when and how these models are updated over time is opaque. Here, we evaluate the March 2023 and June 2023 versions of GPT-3.5 and GPT-4 on several diverse tasks: 1) math problems, 2) sensitive/dangerous questions, 3) opinion surveys, 4) multi-hop knowledge-intensive questions, 5) generating code, 6) US Medical License tests, and 7) visual reasoning. We find that the performance and behavior of both GPT-3.5 and GPT-4 can vary greatly over time. For example, GPT-4 (March 2023) was reasonable at identifyi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.09009","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-18T06:56:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"c4c9097321802a2ca081f6e0104b8a2c4ae58369ef2a7b3514055cb957cf7c96","abstract_canon_sha256":"5de7b612f7a60bcc07de677c5149d01e7195bdd5a916ca5bd63b30b73a69433d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:07:13.480060Z","signature_b64":"vtyeGMlhRUVIqr1UHzh32TRe4ieCJkKITvV7AOVhurb9s8eodPqGrVzl4HLfMPQIkYD+OzMiirICclgsePpTBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eae1f336c40a9d4201490d352d329d9bb3a70cc96e5588b2482993e7b6345775","last_reissued_at":"2026-07-05T07:07:13.479475Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:07:13.479475Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How is ChatGPT's behavior changing over time?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"James Zou, Lingjiao Chen, Matei Zaharia","submitted_at":"2023-07-18T06:56:08Z","abstract_excerpt":"GPT-3.5 and GPT-4 are the two most widely used large language model (LLM) services. However, when and how these models are updated over time is opaque. Here, we evaluate the March 2023 and June 2023 versions of GPT-3.5 and GPT-4 on several diverse tasks: 1) math problems, 2) sensitive/dangerous questions, 3) opinion surveys, 4) multi-hop knowledge-intensive questions, 5) generating code, 6) US Medical License tests, and 7) visual reasoning. We find that the performance and behavior of both GPT-3.5 and GPT-4 can vary greatly over time. For example, GPT-4 (March 2023) was reasonable at identifyi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.09009","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.09009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.09009","created_at":"2026-07-05T07:07:13.479536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.09009v3","created_at":"2026-07-05T07:07:13.479536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.09009","created_at":"2026-07-05T07:07:13.479536+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LQ7GNWEBKOU","created_at":"2026-07-05T07:07:13.479536+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LQ7GNWEBKOUEAKJ","created_at":"2026-07-05T07:07:13.479536+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LQ7GNWE","created_at":"2026-07-05T07:07:13.479536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01421","citing_title":"Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00527","citing_title":"AI Native Games: A Survey and Roadmap","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01391","citing_title":"VISTA: Video Interaction Spatio-Temporal Analysis Benchmark","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25673","citing_title":"Referential Security as a New Paradigm for AI Evaluations","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00603","citing_title":"Toward Agentic Governance: What Shapes LLM-Agent Intervention in Public Forums?","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15326","citing_title":"Analyzing the Presentation, Content, and Utilization of References in LLM-powered Conversational AI Systems","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10253","citing_title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13905","citing_title":"A Non-Destructive Methodological Framework for Modernizing Legacy Clinical Reporting Systems for AI-Driven Pharmacoinformatics: A SAS Case Study","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2406.06608","citing_title":"The Prompt Report: A Systematic Survey of Prompt Engineering Techniques","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2310.11511","citing_title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06365","citing_title":"From Agent Loops to Deterministic Graphs: Execution Lineage for Reproducible AI-Native Work","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01391","citing_title":"VISTA: Video Interaction Spatio-Temporal Analysis Benchmark","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13346","citing_title":"AgentSPEX: An Agent SPecification and EXecution Language","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15409","citing_title":"The Illusion of Equivalence: Systematic FP16 Divergence in KV-Cached Autoregressive Inference","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO","json":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO.json","graph_json":"https://pith.science/api/pith-number/5LQ7GNWEBKOUEAKJBU2S2MU5TO/graph.json","events_json":"https://pith.science/api/pith-number/5LQ7GNWEBKOUEAKJBU2S2MU5TO/events.json","paper":"https://pith.science/paper/5LQ7GNWE"},"agent_actions":{"view_html":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO","download_json":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO.json","view_paper":"https://pith.science/paper/5LQ7GNWE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.09009&json=true","fetch_graph":"https://pith.science/api/pith-number/5LQ7GNWEBKOUEAKJBU2S2MU5TO/graph.json","fetch_events":"https://pith.science/api/pith-number/5LQ7GNWEBKOUEAKJBU2S2MU5TO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO/action/storage_attestation","attest_author":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO/action/author_attestation","sign_citation":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO/action/citation_signature","submit_replication":"https://pith.science/pith/5LQ7GNWEBKOUEAKJBU2S2MU5TO/action/replication_record"}},"created_at":"2026-07-05T07:07:13.479536+00:00","updated_at":"2026-07-05T07:07:13.479536+00:00"}