{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KL57346WQQORO6ZEW7OR4ZJZRI","short_pith_number":"pith:KL57346W","schema_version":"1.0","canonical_sha256":"52fbfdf3d6841d177b24b7dd1e65398a24610804cc4c757ed486789e55caecd4","source":{"kind":"arxiv","id":"2309.08163","version":2},"attestation_state":"computed","paper":{"title":"Self-Assessment Tests are Unreliable Measures of LLM Personality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akshat Gupta, Gopala Anumanchipalli, Xiaoyang Song","submitted_at":"2023-09-15T05:19:39Z","abstract_excerpt":"As large language models (LLM) evolve in their capabilities, various recent studies have tried to quantify their behavior using psychological tools created to study human behavior. One such example is the measurement of \"personality\" of LLMs using self-assessment personality tests developed to measure human personality. Yet almost none of these works verify the applicability of these tests on LLMs. In this paper, we analyze the reliability of LLM personality scores obtained from self-assessment personality tests using two simple experiments. We first introduce the property of prompt sensitivit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.08163","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-15T05:19:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"298b8600250841659ad659bd12633e9111923287a5d39a812fb5fefb3b2a8640","abstract_canon_sha256":"73778cfdab139fac0b63d5d32526762cf7ba00c53e28bfe95056bb66fdd64735"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:39.732898Z","signature_b64":"qVQ0BARAa3A/JHgWlClUp8cOkXFcTePk8Fsb+SGQG5V5QKFYXIJxnIG1VzU5BZgGKVGZQsM2l8BQQy6fnLRZCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52fbfdf3d6841d177b24b7dd1e65398a24610804cc4c757ed486789e55caecd4","last_reissued_at":"2026-07-05T07:29:39.732468Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:39.732468Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Assessment Tests are Unreliable Measures of LLM Personality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akshat Gupta, Gopala Anumanchipalli, Xiaoyang Song","submitted_at":"2023-09-15T05:19:39Z","abstract_excerpt":"As large language models (LLM) evolve in their capabilities, various recent studies have tried to quantify their behavior using psychological tools created to study human behavior. One such example is the measurement of \"personality\" of LLMs using self-assessment personality tests developed to measure human personality. Yet almost none of these works verify the applicability of these tests on LLMs. In this paper, we analyze the reliability of LLM personality scores obtained from self-assessment personality tests using two simple experiments. We first introduce the property of prompt sensitivit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.08163","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.08163/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.08163","created_at":"2026-07-05T07:29:39.732521+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.08163v2","created_at":"2026-07-05T07:29:39.732521+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.08163","created_at":"2026-07-05T07:29:39.732521+00:00"},{"alias_kind":"pith_short_12","alias_value":"KL57346WQQOR","created_at":"2026-07-05T07:29:39.732521+00:00"},{"alias_kind":"pith_short_16","alias_value":"KL57346WQQORO6ZE","created_at":"2026-07-05T07:29:39.732521+00:00"},{"alias_kind":"pith_short_8","alias_value":"KL57346W","created_at":"2026-07-05T07:29:39.732521+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07918","citing_title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09843","citing_title":"An LLM-Native Psychometric Instrument Reveals a Self-Report--Behavior Gap Across 25 Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12730","citing_title":"Rethinking Psychometric Evaluation of LLMs: When and Why Self-Reports Predict Behavior","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2408.09049","citing_title":"Inertia in Moral and Value Judgments of Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2507.11198","citing_title":"Temperature and Persona Shape LLM Agent Consensus With Minimal Accuracy Gains in Qualitative Coding","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI","json":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI.json","graph_json":"https://pith.science/api/pith-number/KL57346WQQORO6ZEW7OR4ZJZRI/graph.json","events_json":"https://pith.science/api/pith-number/KL57346WQQORO6ZEW7OR4ZJZRI/events.json","paper":"https://pith.science/paper/KL57346W"},"agent_actions":{"view_html":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI","download_json":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI.json","view_paper":"https://pith.science/paper/KL57346W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.08163&json=true","fetch_graph":"https://pith.science/api/pith-number/KL57346WQQORO6ZEW7OR4ZJZRI/graph.json","fetch_events":"https://pith.science/api/pith-number/KL57346WQQORO6ZEW7OR4ZJZRI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI/action/storage_attestation","attest_author":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI/action/author_attestation","sign_citation":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI/action/citation_signature","submit_replication":"https://pith.science/pith/KL57346WQQORO6ZEW7OR4ZJZRI/action/replication_record"}},"created_at":"2026-07-05T07:29:39.732521+00:00","updated_at":"2026-07-05T07:29:39.732521+00:00"}