{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ELOFV73FFIYNAYWH7GC6TTI3JP","short_pith_number":"pith:ELOFV73F","schema_version":"1.0","canonical_sha256":"22dc5aff652a30d062c7f985e9cd1b4bf9fc2267ade54ef637165b80da950f90","source":{"kind":"arxiv","id":"2502.19414","version":1},"attestation_state":"computed","paper":{"title":"Can Language Models Falsify? Evaluating Algorithmic Reasoning with Counterexample Creation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.LG","authors_text":"Ameya Prabhu, Jonas Geiping, Matthias Bethge, Ponnurangam Kumaraguru, Shashwat Goel, Shiven Sinha","submitted_at":"2025-02-26T18:58:13Z","abstract_excerpt":"There is growing excitement about the potential of Language Models (LMs) to accelerate scientific discovery. Falsifying hypotheses is key to scientific progress, as it allows claims to be iteratively refined over time. This process requires significant researcher effort, reasoning, and ingenuity. Yet current benchmarks for LMs predominantly assess their ability to generate solutions rather than challenge them. We advocate for developing benchmarks that evaluate this inverse capability - creating counterexamples for subtly incorrect solutions. To demonstrate this approach, we start with the dom"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.19414","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-26T18:58:13Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"7090ef17a7e5a11d178c041ae5a4c8bafbc713b956538034abbaeadebcd50fdf","abstract_canon_sha256":"fa6c4a66aa29d0f8ea8472a40188f0f3220c1870ac24374a1cadfcce907be71b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:31.942219Z","signature_b64":"roU3IRsy/3DqJ2hBc17IojmdGZ2ywSQKvsJpcUTWp4/PbbLHtdM1uQd1yn6HOT1cipVuY/Rq8MDkQTOReH8iBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22dc5aff652a30d062c7f985e9cd1b4bf9fc2267ade54ef637165b80da950f90","last_reissued_at":"2026-07-05T10:20:31.941727Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:31.941727Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Language Models Falsify? Evaluating Algorithmic Reasoning with Counterexample Creation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.LG","authors_text":"Ameya Prabhu, Jonas Geiping, Matthias Bethge, Ponnurangam Kumaraguru, Shashwat Goel, Shiven Sinha","submitted_at":"2025-02-26T18:58:13Z","abstract_excerpt":"There is growing excitement about the potential of Language Models (LMs) to accelerate scientific discovery. Falsifying hypotheses is key to scientific progress, as it allows claims to be iteratively refined over time. This process requires significant researcher effort, reasoning, and ingenuity. Yet current benchmarks for LMs predominantly assess their ability to generate solutions rather than challenge them. We advocate for developing benchmarks that evaluate this inverse capability - creating counterexamples for subtly incorrect solutions. To demonstrate this approach, we start with the dom"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.19414","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.19414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.19414","created_at":"2026-07-05T10:20:31.941788+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.19414v1","created_at":"2026-07-05T10:20:31.941788+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.19414","created_at":"2026-07-05T10:20:31.941788+00:00"},{"alias_kind":"pith_short_12","alias_value":"ELOFV73FFIYN","created_at":"2026-07-05T10:20:31.941788+00:00"},{"alias_kind":"pith_short_16","alias_value":"ELOFV73FFIYNAYWH","created_at":"2026-07-05T10:20:31.941788+00:00"},{"alias_kind":"pith_short_8","alias_value":"ELOFV73F","created_at":"2026-07-05T10:20:31.941788+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP","json":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP.json","graph_json":"https://pith.science/api/pith-number/ELOFV73FFIYNAYWH7GC6TTI3JP/graph.json","events_json":"https://pith.science/api/pith-number/ELOFV73FFIYNAYWH7GC6TTI3JP/events.json","paper":"https://pith.science/paper/ELOFV73F"},"agent_actions":{"view_html":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP","download_json":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP.json","view_paper":"https://pith.science/paper/ELOFV73F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.19414&json=true","fetch_graph":"https://pith.science/api/pith-number/ELOFV73FFIYNAYWH7GC6TTI3JP/graph.json","fetch_events":"https://pith.science/api/pith-number/ELOFV73FFIYNAYWH7GC6TTI3JP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP/action/storage_attestation","attest_author":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP/action/author_attestation","sign_citation":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP/action/citation_signature","submit_replication":"https://pith.science/pith/ELOFV73FFIYNAYWH7GC6TTI3JP/action/replication_record"}},"created_at":"2026-07-05T10:20:31.941788+00:00","updated_at":"2026-07-05T10:20:31.941788+00:00"}