{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MQSX2W5QKM6LEWQULZZJTDIUUK","short_pith_number":"pith:MQSX2W5Q","schema_version":"1.0","canonical_sha256":"64257d5bb0533cb25a145e72998d14a2ae5b34f16f0e6bf8d6bd5bd7792f6299","source":{"kind":"arxiv","id":"2410.10998","version":1},"attestation_state":"computed","paper":{"title":"WILT: A Multi-Turn, Memorization-Robust Inductive Logic Benchmark for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Eryk Banatt, Jonathan Cheng, Skanda Vaidyanath, Tiffany Hwu","submitted_at":"2024-10-14T18:29:13Z","abstract_excerpt":"While large language models have shown impressive capabilities across a wide range of domains, they still encounter significant challenges in reasoning tasks that require gathering evidence over multiple turns and drawing logical conclusions. These challenges present significant obstacles for LLM chat user interfaces, which rely on multi-turn interactions to facilitate effective collaboration. This limitation leads to real-world issues; for example, service chatbots must gather necessary information from customers over multiple turns to diagnose and resolve problems effectively. Despite the mu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10998","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-14T18:29:13Z","cross_cats_sorted":[],"title_canon_sha256":"5ab863708b351543a51e6211c7b9010cfcda155b66e36ad16cf7080a182a4e03","abstract_canon_sha256":"dc71c7f087aa489a2f657510eaf500287d1f29ea6c027e54213065c3042bfdc2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:27.831974Z","signature_b64":"bEEsjqjwUCBYAdhMXU4hI3AYF2T5vvGVOpRw9rsZmxL4pXnpy3PuU+woKnjR2Ob3XE/NRMTX/PaOJ+QEdLxeAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64257d5bb0533cb25a145e72998d14a2ae5b34f16f0e6bf8d6bd5bd7792f6299","last_reissued_at":"2026-07-05T09:20:27.831483Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:27.831483Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WILT: A Multi-Turn, Memorization-Robust Inductive Logic Benchmark for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Eryk Banatt, Jonathan Cheng, Skanda Vaidyanath, Tiffany Hwu","submitted_at":"2024-10-14T18:29:13Z","abstract_excerpt":"While large language models have shown impressive capabilities across a wide range of domains, they still encounter significant challenges in reasoning tasks that require gathering evidence over multiple turns and drawing logical conclusions. These challenges present significant obstacles for LLM chat user interfaces, which rely on multi-turn interactions to facilitate effective collaboration. This limitation leads to real-world issues; for example, service chatbots must gather necessary information from customers over multiple turns to diagnose and resolve problems effectively. Despite the mu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10998","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10998/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10998","created_at":"2026-07-05T09:20:27.831548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10998v1","created_at":"2026-07-05T09:20:27.831548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10998","created_at":"2026-07-05T09:20:27.831548+00:00"},{"alias_kind":"pith_short_12","alias_value":"MQSX2W5QKM6L","created_at":"2026-07-05T09:20:27.831548+00:00"},{"alias_kind":"pith_short_16","alias_value":"MQSX2W5QKM6LEWQU","created_at":"2026-07-05T09:20:27.831548+00:00"},{"alias_kind":"pith_short_8","alias_value":"MQSX2W5Q","created_at":"2026-07-05T09:20:27.831548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23051","citing_title":"Evaluating Temporal Consistency in Multi-Turn Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK","json":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK.json","graph_json":"https://pith.science/api/pith-number/MQSX2W5QKM6LEWQULZZJTDIUUK/graph.json","events_json":"https://pith.science/api/pith-number/MQSX2W5QKM6LEWQULZZJTDIUUK/events.json","paper":"https://pith.science/paper/MQSX2W5Q"},"agent_actions":{"view_html":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK","download_json":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK.json","view_paper":"https://pith.science/paper/MQSX2W5Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10998&json=true","fetch_graph":"https://pith.science/api/pith-number/MQSX2W5QKM6LEWQULZZJTDIUUK/graph.json","fetch_events":"https://pith.science/api/pith-number/MQSX2W5QKM6LEWQULZZJTDIUUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK/action/storage_attestation","attest_author":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK/action/author_attestation","sign_citation":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK/action/citation_signature","submit_replication":"https://pith.science/pith/MQSX2W5QKM6LEWQULZZJTDIUUK/action/replication_record"}},"created_at":"2026-07-05T09:20:27.831548+00:00","updated_at":"2026-07-05T09:20:27.831548+00:00"}