{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RAD6W7JXF3626RC6ZQCNHFC3FR","short_pith_number":"pith:RAD6W7JX","schema_version":"1.0","canonical_sha256":"8807eb7d372efdaf445ecc04d3945b2c5992fb12e395581b847969ef9b189544","source":{"kind":"arxiv","id":"2310.15147","version":2},"attestation_state":"computed","paper":{"title":"S3Eval: A Synthetic, Scalable, Systematic Evaluation Suite for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangyu Lei, Jun Zhao, Kang Liu, Qian Liu, Shizhu He, Yiming Huang","submitted_at":"2023-10-23T17:52:06Z","abstract_excerpt":"The rapid development of Large Language Models (LLMs) has led to great strides in model capabilities like long-context understanding and reasoning. However, as LLMs are able to process longer contexts, it becomes more challenging to evaluate whether they have acquired certain capabilities, since the length of text (e.g., 200K tokens) they can process far exceeds what humans can reliably assess in a reasonable duration. In this paper, we propose using complex synthetic tasks as a proxy evaluation method, and present S3Eval, a Synthetic, Scalable, Systematic evaluation suite for LLMs evaluation."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.15147","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-23T17:52:06Z","cross_cats_sorted":[],"title_canon_sha256":"d3d1e78f027ff2fb02f0bd0e65249af19fdd0c36feec8191f08397f13cf5ce2b","abstract_canon_sha256":"26fd21071e8ff018db8bd26de8a063f3222b989d6f590aed211bdf4e4213ffab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:04.249888Z","signature_b64":"3uvacn32jdjfgfSG0H55r8up/no+gwfd+3f72ViuQyJ4BXe/WH/QFPBj/P/UnTvy0ZJ7iGa368fU8YB4Pb45CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8807eb7d372efdaf445ecc04d3945b2c5992fb12e395581b847969ef9b189544","last_reissued_at":"2026-07-05T08:05:04.249400Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:04.249400Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"S3Eval: A Synthetic, Scalable, Systematic Evaluation Suite for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangyu Lei, Jun Zhao, Kang Liu, Qian Liu, Shizhu He, Yiming Huang","submitted_at":"2023-10-23T17:52:06Z","abstract_excerpt":"The rapid development of Large Language Models (LLMs) has led to great strides in model capabilities like long-context understanding and reasoning. However, as LLMs are able to process longer contexts, it becomes more challenging to evaluate whether they have acquired certain capabilities, since the length of text (e.g., 200K tokens) they can process far exceeds what humans can reliably assess in a reasonable duration. In this paper, we propose using complex synthetic tasks as a proxy evaluation method, and present S3Eval, a Synthetic, Scalable, Systematic evaluation suite for LLMs evaluation."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.15147","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.15147/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.15147","created_at":"2026-07-05T08:05:04.249463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.15147v2","created_at":"2026-07-05T08:05:04.249463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.15147","created_at":"2026-07-05T08:05:04.249463+00:00"},{"alias_kind":"pith_short_12","alias_value":"RAD6W7JXF362","created_at":"2026-07-05T08:05:04.249463+00:00"},{"alias_kind":"pith_short_16","alias_value":"RAD6W7JXF3626RC6","created_at":"2026-07-05T08:05:04.249463+00:00"},{"alias_kind":"pith_short_8","alias_value":"RAD6W7JX","created_at":"2026-07-05T08:05:04.249463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.11584","citing_title":"The Landscape of Emerging AI Agent Architectures for Reasoning, Planning, and Tool Calling: A Survey","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24536","citing_title":"Generating Place-Based Compromises Between Two Points of View","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR","json":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR.json","graph_json":"https://pith.science/api/pith-number/RAD6W7JXF3626RC6ZQCNHFC3FR/graph.json","events_json":"https://pith.science/api/pith-number/RAD6W7JXF3626RC6ZQCNHFC3FR/events.json","paper":"https://pith.science/paper/RAD6W7JX"},"agent_actions":{"view_html":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR","download_json":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR.json","view_paper":"https://pith.science/paper/RAD6W7JX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.15147&json=true","fetch_graph":"https://pith.science/api/pith-number/RAD6W7JXF3626RC6ZQCNHFC3FR/graph.json","fetch_events":"https://pith.science/api/pith-number/RAD6W7JXF3626RC6ZQCNHFC3FR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR/action/storage_attestation","attest_author":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR/action/author_attestation","sign_citation":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR/action/citation_signature","submit_replication":"https://pith.science/pith/RAD6W7JXF3626RC6ZQCNHFC3FR/action/replication_record"}},"created_at":"2026-07-05T08:05:04.249463+00:00","updated_at":"2026-07-05T08:05:04.249463+00:00"}