{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VFS7P4RE7HCG552IZB6J3BDBZ7","short_pith_number":"pith:VFS7P4RE","schema_version":"1.0","canonical_sha256":"a965f7f224f9c46ef748c87c9d8461cfebca259861be89e03c617d5ec87e6e2c","source":{"kind":"arxiv","id":"2309.17167","version":3},"attestation_state":"computed","paper":{"title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Diyi Yang, Jiaao Chen, Jindong Wang, Kaijie Zhu, Neil Zhenqiang Gong, Xing Xie","submitted_at":"2023-09-29T12:04:14Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable performance in various evaluation benchmarks. However, concerns are raised about potential data contamination in their considerable volume of training corpus. Moreover, the static nature and fixed complexity of current benchmarks may inadequately gauge the advancing capabilities of LLMs. In this paper, we introduce DyVal, a general and flexible protocol for dynamic evaluation of LLMs. Based on our framework, we build graph-informed DyVal by leveraging the structural advantage of directed acyclic graphs to dynamically generate evaluation sam"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.17167","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-29T12:04:14Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"aea1f233d4e99aef0aab6fbc893c7b84cf545718f961f7db7a60b475f6c1bf5d","abstract_canon_sha256":"77c6a4950a44cb36ec4f1478c126940d8451ba6b0999678c420191a6376fcdc1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:55:59.722457Z","signature_b64":"E0CZNQtJvc34UisHxntM1XkU01aXkrED01X5Q8YBsp0/rmhiF6z2bXGOOxtNW/+oRDsYwHbTf66qwqRczO4CBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a965f7f224f9c46ef748c87c9d8461cfebca259861be89e03c617d5ec87e6e2c","last_reissued_at":"2026-07-05T07:55:59.721946Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:55:59.721946Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Diyi Yang, Jiaao Chen, Jindong Wang, Kaijie Zhu, Neil Zhenqiang Gong, Xing Xie","submitted_at":"2023-09-29T12:04:14Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable performance in various evaluation benchmarks. However, concerns are raised about potential data contamination in their considerable volume of training corpus. Moreover, the static nature and fixed complexity of current benchmarks may inadequately gauge the advancing capabilities of LLMs. In this paper, we introduce DyVal, a general and flexible protocol for dynamic evaluation of LLMs. Based on our framework, we build graph-informed DyVal by leveraging the structural advantage of directed acyclic graphs to dynamically generate evaluation sam"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.17167","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.17167/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.17167","created_at":"2026-07-05T07:55:59.722012+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.17167v3","created_at":"2026-07-05T07:55:59.722012+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.17167","created_at":"2026-07-05T07:55:59.722012+00:00"},{"alias_kind":"pith_short_12","alias_value":"VFS7P4RE7HCG","created_at":"2026-07-05T07:55:59.722012+00:00"},{"alias_kind":"pith_short_16","alias_value":"VFS7P4RE7HCG552I","created_at":"2026-07-05T07:55:59.722012+00:00"},{"alias_kind":"pith_short_8","alias_value":"VFS7P4RE","created_at":"2026-07-05T07:55:59.722012+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02141","citing_title":"A$^{2}$utoLPBench: An Auto-Generated, Agent-Friendly LP Benchmark via Inverse-KKT Construction","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22564","citing_title":"SynAE: A Framework for Measuring the Quality of Synthetic Data for Tool-Calling Agent Evaluations","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15865","citing_title":"From Text to DSL: Evaluating Grammar-Based Model Generation Using Open LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2310.19852","citing_title":"AI Alignment: A Comprehensive Survey","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2404.11584","citing_title":"The Landscape of Emerging AI Agent Architectures for Reasoning, Planning, and Tool Calling: A Survey","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08904","citing_title":"OPT-BENCH: Evaluating the Iterative Self-Optimization of LLM Agents in Large-Scale Search Spaces","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17842","citing_title":"QuickScope: Certifying Hard Questions in Dynamic LLM Benchmarks","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7","json":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7.json","graph_json":"https://pith.science/api/pith-number/VFS7P4RE7HCG552IZB6J3BDBZ7/graph.json","events_json":"https://pith.science/api/pith-number/VFS7P4RE7HCG552IZB6J3BDBZ7/events.json","paper":"https://pith.science/paper/VFS7P4RE"},"agent_actions":{"view_html":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7","download_json":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7.json","view_paper":"https://pith.science/paper/VFS7P4RE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.17167&json=true","fetch_graph":"https://pith.science/api/pith-number/VFS7P4RE7HCG552IZB6J3BDBZ7/graph.json","fetch_events":"https://pith.science/api/pith-number/VFS7P4RE7HCG552IZB6J3BDBZ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7/action/storage_attestation","attest_author":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7/action/author_attestation","sign_citation":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7/action/citation_signature","submit_replication":"https://pith.science/pith/VFS7P4RE7HCG552IZB6J3BDBZ7/action/replication_record"}},"created_at":"2026-07-05T07:55:59.722012+00:00","updated_at":"2026-07-05T07:55:59.722012+00:00"}