{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NL2LCGWDJVUJXPUTN6CAC42VED","short_pith_number":"pith:NL2LCGWD","schema_version":"1.0","canonical_sha256":"6af4b11ac34d689bbe936f8401735520f6be9be80a0478068338117df194b024","source":{"kind":"arxiv","id":"2507.21476","version":1},"attestation_state":"computed","paper":{"title":"Which LLMs Get the Joke? Probing Non-STEM Reasoning Abilities with HumorBench","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bob Mankoff, Jiayi Chen, Jifan Zhang, Lalit Jain, Pine S.L. Dysart-Bricken, Reuben Narad, Robert Nowak, Siddharth Suresh","submitted_at":"2025-07-29T03:44:43Z","abstract_excerpt":"We present HumorBench, a benchmark designed to evaluate large language models' (LLMs) ability to reason about and explain sophisticated humor in cartoon captions. As reasoning models increasingly saturate existing benchmarks in mathematics and science, novel and challenging evaluations of model intelligence beyond STEM domains are essential. Reasoning is fundamentally involved in text-based humor comprehension, requiring the identification of connections between concepts in cartoons/captions and external cultural references, wordplays, and other mechanisms. HumorBench includes approximately 30"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.21476","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-29T03:44:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1aad0d60e3a42ca2b7dae58bcb0fa7fc1a3681ef8e3e8e3bf08ee7d84becbd0c","abstract_canon_sha256":"cd645fff5f1abee5f8f3155cbbf5116826b3fde41a51c5f15a832dc60e79a75f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:00.747171Z","signature_b64":"7om+F3GmkExiyBq7zxpPzjGNxTQQpewp3XwD8qMtc9b+OSYHuvYOt8BxYgho0aaBBiWUd4R81nzRN+cVbrNcCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6af4b11ac34d689bbe936f8401735520f6be9be80a0478068338117df194b024","last_reissued_at":"2026-07-05T11:45:00.746687Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:00.746687Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Which LLMs Get the Joke? Probing Non-STEM Reasoning Abilities with HumorBench","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bob Mankoff, Jiayi Chen, Jifan Zhang, Lalit Jain, Pine S.L. Dysart-Bricken, Reuben Narad, Robert Nowak, Siddharth Suresh","submitted_at":"2025-07-29T03:44:43Z","abstract_excerpt":"We present HumorBench, a benchmark designed to evaluate large language models' (LLMs) ability to reason about and explain sophisticated humor in cartoon captions. As reasoning models increasingly saturate existing benchmarks in mathematics and science, novel and challenging evaluations of model intelligence beyond STEM domains are essential. Reasoning is fundamentally involved in text-based humor comprehension, requiring the identification of connections between concepts in cartoons/captions and external cultural references, wordplays, and other mechanisms. HumorBench includes approximately 30"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.21476","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.21476/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.21476","created_at":"2026-07-05T11:45:00.746750+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.21476v1","created_at":"2026-07-05T11:45:00.746750+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.21476","created_at":"2026-07-05T11:45:00.746750+00:00"},{"alias_kind":"pith_short_12","alias_value":"NL2LCGWDJVUJ","created_at":"2026-07-05T11:45:00.746750+00:00"},{"alias_kind":"pith_short_16","alias_value":"NL2LCGWDJVUJXPUT","created_at":"2026-07-05T11:45:00.746750+00:00"},{"alias_kind":"pith_short_8","alias_value":"NL2LCGWD","created_at":"2026-07-05T11:45:00.746750+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19786","citing_title":"HumorRank: A Tournament-Based Leaderboard for Evaluating Humor Generation in Large Language Models","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED","json":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED.json","graph_json":"https://pith.science/api/pith-number/NL2LCGWDJVUJXPUTN6CAC42VED/graph.json","events_json":"https://pith.science/api/pith-number/NL2LCGWDJVUJXPUTN6CAC42VED/events.json","paper":"https://pith.science/paper/NL2LCGWD"},"agent_actions":{"view_html":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED","download_json":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED.json","view_paper":"https://pith.science/paper/NL2LCGWD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.21476&json=true","fetch_graph":"https://pith.science/api/pith-number/NL2LCGWDJVUJXPUTN6CAC42VED/graph.json","fetch_events":"https://pith.science/api/pith-number/NL2LCGWDJVUJXPUTN6CAC42VED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED/action/storage_attestation","attest_author":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED/action/author_attestation","sign_citation":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED/action/citation_signature","submit_replication":"https://pith.science/pith/NL2LCGWDJVUJXPUTN6CAC42VED/action/replication_record"}},"created_at":"2026-07-05T11:45:00.746750+00:00","updated_at":"2026-07-05T11:45:00.746750+00:00"}