{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5Q4O7EJEOEOSHEKUXM5A5K3U6Q","short_pith_number":"pith:5Q4O7EJE","schema_version":"1.0","canonical_sha256":"ec38ef9124711d239154bb3a0eab74f427e055269194655165aa88ea3398fb4b","source":{"kind":"arxiv","id":"2311.05297","version":2},"attestation_state":"computed","paper":{"title":"Challenging the Validity of Personality Tests for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Augustin Kelava, Florian E. Dorner, Samira Samadi, Tom S\\\"uhr","submitted_at":"2023-11-09T11:54:01Z","abstract_excerpt":"With large language models (LLMs) like GPT-4 appearing to behave increasingly human-like in text-based interactions, it has become popular to attempt to evaluate personality traits of LLMs using questionnaires originally developed for humans. While reusing measures is a resource-efficient way to evaluate LLMs, careful adaptations are usually required to ensure that assessment results are valid even across human subpopulations. In this work, we provide evidence that LLMs' responses to personality tests systematically deviate from human responses, implying that the results of these tests cannot "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.05297","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-09T11:54:01Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"33f9336fe92c78fb41c6a946da89c584f912d727ea0bc3ffd4d3a74605268451","abstract_canon_sha256":"f9262f50f55d253d43ca3c80d16e8af7bf442f8c40cf65468e66b9ae33683dab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:43.784833Z","signature_b64":"FdgDEQSjOT5hp+CdAxgJLY3zzaJg7a7STVnlakfH/8keJ63Qvv9BsvIrQgrC3pe2AYo/cQpow6aSsH/yCFtmDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec38ef9124711d239154bb3a0eab74f427e055269194655165aa88ea3398fb4b","last_reissued_at":"2026-07-05T08:27:43.784284Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:43.784284Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Challenging the Validity of Personality Tests for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Augustin Kelava, Florian E. Dorner, Samira Samadi, Tom S\\\"uhr","submitted_at":"2023-11-09T11:54:01Z","abstract_excerpt":"With large language models (LLMs) like GPT-4 appearing to behave increasingly human-like in text-based interactions, it has become popular to attempt to evaluate personality traits of LLMs using questionnaires originally developed for humans. While reusing measures is a resource-efficient way to evaluate LLMs, careful adaptations are usually required to ensure that assessment results are valid even across human subpopulations. In this work, we provide evidence that LLMs' responses to personality tests systematically deviate from human responses, implying that the results of these tests cannot "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.05297","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.05297/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.05297","created_at":"2026-07-05T08:27:43.784351+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.05297v2","created_at":"2026-07-05T08:27:43.784351+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.05297","created_at":"2026-07-05T08:27:43.784351+00:00"},{"alias_kind":"pith_short_12","alias_value":"5Q4O7EJEOEOS","created_at":"2026-07-05T08:27:43.784351+00:00"},{"alias_kind":"pith_short_16","alias_value":"5Q4O7EJEOEOSHEKU","created_at":"2026-07-05T08:27:43.784351+00:00"},{"alias_kind":"pith_short_8","alias_value":"5Q4O7EJE","created_at":"2026-07-05T08:27:43.784351+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20603","citing_title":"A Survey of Large Language Models for Perception and Measurement of Human Psychology","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2408.09049","citing_title":"Inertia in Moral and Value Judgments of Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05080","citing_title":"The Pinocchio Dimension: Phenomenality of Experience as the Primary Axis of LLM Psychometric Differences","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10733","citing_title":"Too Nice to Tell the Truth: Quantifying Agreeableness-Driven Sycophancy in Role-Playing Language Models","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q","json":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q.json","graph_json":"https://pith.science/api/pith-number/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/graph.json","events_json":"https://pith.science/api/pith-number/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/events.json","paper":"https://pith.science/paper/5Q4O7EJE"},"agent_actions":{"view_html":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q","download_json":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q.json","view_paper":"https://pith.science/paper/5Q4O7EJE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.05297&json=true","fetch_graph":"https://pith.science/api/pith-number/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/graph.json","fetch_events":"https://pith.science/api/pith-number/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/action/storage_attestation","attest_author":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/action/author_attestation","sign_citation":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/action/citation_signature","submit_replication":"https://pith.science/pith/5Q4O7EJEOEOSHEKUXM5A5K3U6Q/action/replication_record"}},"created_at":"2026-07-05T08:27:43.784351+00:00","updated_at":"2026-07-05T08:27:43.784351+00:00"}