{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LGLYFYS3N5MZN4Z65FGSQVXAWY","short_pith_number":"pith:LGLYFYS3","schema_version":"1.0","canonical_sha256":"599782e25b6f5996f33ee94d2856e0b60ccb1c84392494a443bf74c5f601dce6","source":{"kind":"arxiv","id":"2502.19187","version":2},"attestation_state":"computed","paper":{"title":"BIG-Bench Extra Hard","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bahare Fatemi, Chrysovalantis Anastasiou, Disha Jindal, Gladys Tyen, Hritik Bansal, John Palowitch, Kate Olszewska, Lalit K. Jain, Mehran Kazemi, Nishanth Dikkala, Orhan Firat, Peter Chen, Quoc V. Le, Sanket Vaibhav Mehta, Silvia Chiappa, Uri Shalit, Vinh Q. Tran, Virginia Aglietti, Xin Liu, Yi Tay","submitted_at":"2025-02-26T14:50:50Z","abstract_excerpt":"Large language models (LLMs) are increasingly deployed in everyday applications, demanding robust general reasoning capabilities and diverse reasoning skillset. However, current LLM reasoning benchmarks predominantly focus on mathematical and coding abilities, leaving a gap in evaluating broader reasoning proficiencies. One particular exception is the BIG-Bench dataset, which has served as a crucial benchmark for evaluating the general reasoning capabilities of LLMs, thanks to its diverse set of challenging tasks that allowed for a comprehensive assessment of general reasoning across various s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.19187","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-26T14:50:50Z","cross_cats_sorted":[],"title_canon_sha256":"dc9f2ff5c541fa297b52f4390572c0ae609ebc652b8c2c52889e86e4286ff14a","abstract_canon_sha256":"4a7ffd350814a22ac3583bdaefc87b3d423ae0a07e3c27de4f5dba176ffc697d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:05.976383Z","signature_b64":"acheoGrJIExWnFkNwTMmVi2/PGLXNeRHsw4jL+NkyA3CpOB071KLDfi+3nwpukHGBCacz0RTKinV6lJZQpYfBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"599782e25b6f5996f33ee94d2856e0b60ccb1c84392494a443bf74c5f601dce6","last_reissued_at":"2026-07-05T10:59:05.975888Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:05.975888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BIG-Bench Extra Hard","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bahare Fatemi, Chrysovalantis Anastasiou, Disha Jindal, Gladys Tyen, Hritik Bansal, John Palowitch, Kate Olszewska, Lalit K. Jain, Mehran Kazemi, Nishanth Dikkala, Orhan Firat, Peter Chen, Quoc V. Le, Sanket Vaibhav Mehta, Silvia Chiappa, Uri Shalit, Vinh Q. Tran, Virginia Aglietti, Xin Liu, Yi Tay","submitted_at":"2025-02-26T14:50:50Z","abstract_excerpt":"Large language models (LLMs) are increasingly deployed in everyday applications, demanding robust general reasoning capabilities and diverse reasoning skillset. However, current LLM reasoning benchmarks predominantly focus on mathematical and coding abilities, leaving a gap in evaluating broader reasoning proficiencies. One particular exception is the BIG-Bench dataset, which has served as a crucial benchmark for evaluating the general reasoning capabilities of LLMs, thanks to its diverse set of challenging tasks that allowed for a comprehensive assessment of general reasoning across various s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.19187","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.19187/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.19187","created_at":"2026-07-05T10:59:05.975946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.19187v2","created_at":"2026-07-05T10:59:05.975946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.19187","created_at":"2026-07-05T10:59:05.975946+00:00"},{"alias_kind":"pith_short_12","alias_value":"LGLYFYS3N5MZ","created_at":"2026-07-05T10:59:05.975946+00:00"},{"alias_kind":"pith_short_16","alias_value":"LGLYFYS3N5MZN4Z6","created_at":"2026-07-05T10:59:05.975946+00:00"},{"alias_kind":"pith_short_8","alias_value":"LGLYFYS3","created_at":"2026-07-05T10:59:05.975946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15079","citing_title":"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale","ref_index":196,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03800","citing_title":"Trading Human Curation for Synthetic Augmentation in RLVR","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30774","citing_title":"What Drives Interactive Improvement from Feedback?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2508.08636","citing_title":"InternBootcamp Technical Report: Boosting LLM Reasoning with Verifiable Task Scaling","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2507.10722","citing_title":"Bridging Brains and Machines: A Unified Frontier in Neuroscience, Artificial Intelligence, and Neuromorphic Systems","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12549","citing_title":"The Serial Scaling Hypothesis","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23009","citing_title":"Position: Stop Evaluating AI with Human Tests, Develop Principled, AI-specific Tests instead","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14420","citing_title":"Instructions are all you need: Self-supervised Reinforcement Learning for Instruction Following","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.09907","citing_title":"Learning to Pose Problems: Reasoning-Driven and Solver-Adaptive Data Synthesis","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19678","citing_title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07593","citing_title":"Too long; didn't solve","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16646","citing_title":"Agentic Frameworks for Reasoning Tasks: An Empirical Study","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05851","citing_title":"Hypothesis generation and updating in large language models","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY","json":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY.json","graph_json":"https://pith.science/api/pith-number/LGLYFYS3N5MZN4Z65FGSQVXAWY/graph.json","events_json":"https://pith.science/api/pith-number/LGLYFYS3N5MZN4Z65FGSQVXAWY/events.json","paper":"https://pith.science/paper/LGLYFYS3"},"agent_actions":{"view_html":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY","download_json":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY.json","view_paper":"https://pith.science/paper/LGLYFYS3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.19187&json=true","fetch_graph":"https://pith.science/api/pith-number/LGLYFYS3N5MZN4Z65FGSQVXAWY/graph.json","fetch_events":"https://pith.science/api/pith-number/LGLYFYS3N5MZN4Z65FGSQVXAWY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY/action/storage_attestation","attest_author":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY/action/author_attestation","sign_citation":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY/action/citation_signature","submit_replication":"https://pith.science/pith/LGLYFYS3N5MZN4Z65FGSQVXAWY/action/replication_record"}},"created_at":"2026-07-05T10:59:05.975946+00:00","updated_at":"2026-07-05T10:59:05.975946+00:00"}