{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4YFXFEJTUPH3QCFJBGKBHYHAF5","short_pith_number":"pith:4YFXFEJT","schema_version":"1.0","canonical_sha256":"e60b729133a3cfb808a9099413e0e02f6e03aadae92399d72d2a949bab6aa9f3","source":{"kind":"arxiv","id":"2502.14678","version":1},"attestation_state":"computed","paper":{"title":"How to Get Your LLM to Generate Challenging Problems for Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arkil Patel, Dzmitry Bahdanau, Siva Reddy","submitted_at":"2025-02-20T16:09:55Z","abstract_excerpt":"The pace of evolution of Large Language Models (LLMs) necessitates new approaches for rigorous and comprehensive evaluation. Traditional human annotation is increasingly impracticable due to the complexities and costs involved in generating high-quality, challenging problems. In this work, we introduce CHASE, a unified framework to synthetically generate challenging problems using LLMs without human involvement. For a given task, our approach builds a hard problem in a bottom-up manner from simpler components. Moreover, our framework decomposes the generation process into independently verifia"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14678","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-20T16:09:55Z","cross_cats_sorted":[],"title_canon_sha256":"6159de472dadafdf389a442661006c2dd3be13089b20d014b5c18b16fcf48c06","abstract_canon_sha256":"c8db1c285a81761aae18bf5fb7a6eb1a18fdf59f9cc8d71627ffb297402e2a37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:30.990050Z","signature_b64":"d7id1GAg/Q4tKlLL5v0Lqukf4BWVQcfWO37H98s2gTM1u9lZcfY0jWwTBv67qRG9j9lDqIRHlEU62SshUSOmCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e60b729133a3cfb808a9099413e0e02f6e03aadae92399d72d2a949bab6aa9f3","last_reissued_at":"2026-07-05T10:17:30.989610Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:30.989610Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to Get Your LLM to Generate Challenging Problems for Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arkil Patel, Dzmitry Bahdanau, Siva Reddy","submitted_at":"2025-02-20T16:09:55Z","abstract_excerpt":"The pace of evolution of Large Language Models (LLMs) necessitates new approaches for rigorous and comprehensive evaluation. Traditional human annotation is increasingly impracticable due to the complexities and costs involved in generating high-quality, challenging problems. In this work, we introduce CHASE, a unified framework to synthetically generate challenging problems using LLMs without human involvement. For a given task, our approach builds a hard problem in a bottom-up manner from simpler components. Moreover, our framework decomposes the generation process into independently verifia"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14678","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14678/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14678","created_at":"2026-07-05T10:17:30.989672+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14678v1","created_at":"2026-07-05T10:17:30.989672+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14678","created_at":"2026-07-05T10:17:30.989672+00:00"},{"alias_kind":"pith_short_12","alias_value":"4YFXFEJTUPH3","created_at":"2026-07-05T10:17:30.989672+00:00"},{"alias_kind":"pith_short_16","alias_value":"4YFXFEJTUPH3QCFJ","created_at":"2026-07-05T10:17:30.989672+00:00"},{"alias_kind":"pith_short_8","alias_value":"4YFXFEJT","created_at":"2026-07-05T10:17:30.989672+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05464","citing_title":"Step-by-Step Optimization-like Reasoning in LLMs over Expanding Search Spaces","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03650","citing_title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24544","citing_title":"STELLAR-E: a Synthetic, Tailored, End-to-end LLM Application Rigorous Evaluator","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5","json":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5.json","graph_json":"https://pith.science/api/pith-number/4YFXFEJTUPH3QCFJBGKBHYHAF5/graph.json","events_json":"https://pith.science/api/pith-number/4YFXFEJTUPH3QCFJBGKBHYHAF5/events.json","paper":"https://pith.science/paper/4YFXFEJT"},"agent_actions":{"view_html":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5","download_json":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5.json","view_paper":"https://pith.science/paper/4YFXFEJT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14678&json=true","fetch_graph":"https://pith.science/api/pith-number/4YFXFEJTUPH3QCFJBGKBHYHAF5/graph.json","fetch_events":"https://pith.science/api/pith-number/4YFXFEJTUPH3QCFJBGKBHYHAF5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5/action/storage_attestation","attest_author":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5/action/author_attestation","sign_citation":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5/action/citation_signature","submit_replication":"https://pith.science/pith/4YFXFEJTUPH3QCFJBGKBHYHAF5/action/replication_record"}},"created_at":"2026-07-05T10:17:30.989672+00:00","updated_at":"2026-07-05T10:17:30.989672+00:00"}