{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RGXH2KSSJHHAP3LBZT7SUA4Z6U","short_pith_number":"pith:RGXH2KSS","schema_version":"1.0","canonical_sha256":"89ae7d2a5249ce07ed61ccff2a0399f502e211fb7d7c8482835225ca597c55d9","source":{"kind":"arxiv","id":"2310.19736","version":3},"attestation_state":"computed","paper":{"title":"Evaluating Large Language Models: A Comprehensive Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bojian Xiong, Chuang Liu, Dan Shi, Deyi Xiong, Jiaxuan Li, Linhao Yu, Renren Jin, Supryadi, Yan Liu, Yufei Huang, Zishan Guo","submitted_at":"2023-10-30T17:00:52Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities across a broad spectrum of tasks. They have attracted significant attention and been deployed in numerous downstream applications. Nevertheless, akin to a double-edged sword, LLMs also present potential risks. They could suffer from private data leaks or yield inappropriate, harmful, or misleading content. Additionally, the rapid progress of LLMs raises concerns about the potential emergence of superintelligent systems without adequate safeguards. To effectively capitalize on LLM capacities as well as ensure their safe and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.19736","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-30T17:00:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4835ade77e70140dfff8b6abd4d7cf9a5ccbd09f6314a41347a2d3734c3b1146","abstract_canon_sha256":"0444374a702bd843fe081d54cc5761b1b69680681d2f0a252185d6b654ea4782"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:16:48.638659Z","signature_b64":"8yweOFuMtjRlovG1T540TswN7/UiMrN/dix5ScOsWppw15HcfPirc650kBy4vj54L1BPTg0bh/yoO2U117JjAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89ae7d2a5249ce07ed61ccff2a0399f502e211fb7d7c8482835225ca597c55d9","last_reissued_at":"2026-07-05T07:16:48.638172Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:16:48.638172Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Large Language Models: A Comprehensive Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bojian Xiong, Chuang Liu, Dan Shi, Deyi Xiong, Jiaxuan Li, Linhao Yu, Renren Jin, Supryadi, Yan Liu, Yufei Huang, Zishan Guo","submitted_at":"2023-10-30T17:00:52Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities across a broad spectrum of tasks. They have attracted significant attention and been deployed in numerous downstream applications. Nevertheless, akin to a double-edged sword, LLMs also present potential risks. They could suffer from private data leaks or yield inappropriate, harmful, or misleading content. Additionally, the rapid progress of LLMs raises concerns about the potential emergence of superintelligent systems without adequate safeguards. To effectively capitalize on LLM capacities as well as ensure their safe and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19736","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19736/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.19736","created_at":"2026-07-05T07:16:48.638232+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.19736v3","created_at":"2026-07-05T07:16:48.638232+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19736","created_at":"2026-07-05T07:16:48.638232+00:00"},{"alias_kind":"pith_short_12","alias_value":"RGXH2KSSJHHA","created_at":"2026-07-05T07:16:48.638232+00:00"},{"alias_kind":"pith_short_16","alias_value":"RGXH2KSSJHHAP3LB","created_at":"2026-07-05T07:16:48.638232+00:00"},{"alias_kind":"pith_short_8","alias_value":"RGXH2KSS","created_at":"2026-07-05T07:16:48.638232+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22329","citing_title":"BabelJudge: Measuring LLM-as-a-Judge Reliability Across Languages and Agent Trajectories","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21939","citing_title":"Beyond Value Benchmarks: Measuring Value-Structure Alignment in Large Language Models via Symmetric Q-Sorts","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2501.09775","citing_title":"Multiple Choice Questions: Reasoning Makes Large Language Models (LLMs) More Self-Confident, Especially When They are Wrong","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15761","citing_title":"A Unified Perturbation Framework for Analyzing Leaderboard Stability and Manipulation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17170","citing_title":"When LLM Judges Inflate Scores: Exploring Overrating in Relevance Assessment","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12055","citing_title":"Do Language Models Encode Knowledge of Linguistic Constraint Violations?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12055","citing_title":"Do Language Models Encode Knowledge of Linguistic Constraint Violations?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10851","citing_title":"The Generalized Turing Test: A Foundation for Comparing Intelligence","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07883","citing_title":"Beyond \"I cannot fulfill this request\": Alleviating Rigid Rejection in LLMs via Label Enhancement","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U","json":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U.json","graph_json":"https://pith.science/api/pith-number/RGXH2KSSJHHAP3LBZT7SUA4Z6U/graph.json","events_json":"https://pith.science/api/pith-number/RGXH2KSSJHHAP3LBZT7SUA4Z6U/events.json","paper":"https://pith.science/paper/RGXH2KSS"},"agent_actions":{"view_html":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U","download_json":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U.json","view_paper":"https://pith.science/paper/RGXH2KSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.19736&json=true","fetch_graph":"https://pith.science/api/pith-number/RGXH2KSSJHHAP3LBZT7SUA4Z6U/graph.json","fetch_events":"https://pith.science/api/pith-number/RGXH2KSSJHHAP3LBZT7SUA4Z6U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U/action/storage_attestation","attest_author":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U/action/author_attestation","sign_citation":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U/action/citation_signature","submit_replication":"https://pith.science/pith/RGXH2KSSJHHAP3LBZT7SUA4Z6U/action/replication_record"}},"created_at":"2026-07-05T07:16:48.638232+00:00","updated_at":"2026-07-05T07:16:48.638232+00:00"}