{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:54SYME3VH44AE757GGVVX63ZSI","short_pith_number":"pith:54SYME3V","schema_version":"1.0","canonical_sha256":"ef258613753f38027fbf31ab5bfb799213d83668cf284118b4d035b54fd1cb25","source":{"kind":"arxiv","id":"2410.03608","version":1},"attestation_state":"computed","paper":{"title":"TICKing All the Boxes: Generated Checklists Improve LLM Evaluation and Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alex Wang, Dennis Aumiller, Jakob Foerster, Jonathan Cook, Tim Rockt\\\"aschel","submitted_at":"2024-10-04T17:09:08Z","abstract_excerpt":"Given the widespread adoption and usage of Large Language Models (LLMs), it is crucial to have flexible and interpretable evaluations of their instruction-following ability. Preference judgments between model outputs have become the de facto evaluation standard, despite distilling complex, multi-faceted preferences into a single ranking. Furthermore, as human annotation is slow and costly, LLMs are increasingly used to make these judgments, at the expense of reliability and interpretability. In this work, we propose TICK (Targeted Instruct-evaluation with ChecKlists), a fully automated, interp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03608","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-04T17:09:08Z","cross_cats_sorted":["cs.CL","cs.HC","cs.LG"],"title_canon_sha256":"416be20ebf75b39b0e995b0a535ec630cb94a29daea3e8f5dd20574136b9cd14","abstract_canon_sha256":"4716ed3252e671d54d2e15322640d93b952053da233cee105918ddb759bf9fc9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:03.600956Z","signature_b64":"ed2esa/Ne73RF1h7IwtuwHYK6A9yQ0H+UruM1ykwWf9j876WaunX47Qpur9MIOMPXpnr8K6LtI7J5MA+JrjLBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef258613753f38027fbf31ab5bfb799213d83668cf284118b4d035b54fd1cb25","last_reissued_at":"2026-07-05T09:16:03.600541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:03.600541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TICKing All the Boxes: Generated Checklists Improve LLM Evaluation and Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alex Wang, Dennis Aumiller, Jakob Foerster, Jonathan Cook, Tim Rockt\\\"aschel","submitted_at":"2024-10-04T17:09:08Z","abstract_excerpt":"Given the widespread adoption and usage of Large Language Models (LLMs), it is crucial to have flexible and interpretable evaluations of their instruction-following ability. Preference judgments between model outputs have become the de facto evaluation standard, despite distilling complex, multi-faceted preferences into a single ranking. Furthermore, as human annotation is slow and costly, LLMs are increasingly used to make these judgments, at the expense of reliability and interpretability. In this work, we propose TICK (Targeted Instruct-evaluation with ChecKlists), a fully automated, interp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03608","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03608/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03608","created_at":"2026-07-05T09:16:03.600598+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03608v1","created_at":"2026-07-05T09:16:03.600598+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03608","created_at":"2026-07-05T09:16:03.600598+00:00"},{"alias_kind":"pith_short_12","alias_value":"54SYME3VH44A","created_at":"2026-07-05T09:16:03.600598+00:00"},{"alias_kind":"pith_short_16","alias_value":"54SYME3VH44AE757","created_at":"2026-07-05T09:16:03.600598+00:00"},{"alias_kind":"pith_short_8","alias_value":"54SYME3V","created_at":"2026-07-05T09:16:03.600598+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08268","citing_title":"Different Teachers, Different Capabilities: Sub-1B On-Device Distillation for Structured Text Enrichment","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2605.27914","citing_title":"Does Capability Transfer to Subjective Behavior -- and Would Our Instruments Tell Us? A Self-Evolving, Trust-by-Construction Evaluation Paradigm","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI","json":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI.json","graph_json":"https://pith.science/api/pith-number/54SYME3VH44AE757GGVVX63ZSI/graph.json","events_json":"https://pith.science/api/pith-number/54SYME3VH44AE757GGVVX63ZSI/events.json","paper":"https://pith.science/paper/54SYME3V"},"agent_actions":{"view_html":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI","download_json":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI.json","view_paper":"https://pith.science/paper/54SYME3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03608&json=true","fetch_graph":"https://pith.science/api/pith-number/54SYME3VH44AE757GGVVX63ZSI/graph.json","fetch_events":"https://pith.science/api/pith-number/54SYME3VH44AE757GGVVX63ZSI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI/action/storage_attestation","attest_author":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI/action/author_attestation","sign_citation":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI/action/citation_signature","submit_replication":"https://pith.science/pith/54SYME3VH44AE757GGVVX63ZSI/action/replication_record"}},"created_at":"2026-07-05T09:16:03.600598+00:00","updated_at":"2026-07-05T09:16:03.600598+00:00"}