{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2TYANLBB22O7V633HFDQX6KLPA","short_pith_number":"pith:2TYANLBB","schema_version":"1.0","canonical_sha256":"d4f006ac21d69dfafb7b39470bf94b780e66950758d784641ab3c7bb0caa3565","source":{"kind":"arxiv","id":"2507.17476","version":1},"attestation_state":"computed","paper":{"title":"MultiNRC: A Challenging and Native Multilingual Reasoning Evaluation Benchmark for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander R. Fabbri, Bing Liu, Chen Xing, Dean Lee, Diego Mares, Ernesto Hernandez, Jorge Flores, Meher Mankikar","submitted_at":"2025-07-23T12:56:31Z","abstract_excerpt":"Although recent Large Language Models (LLMs) have shown rapid improvement on reasoning benchmarks in English, the evaluation of such LLMs' multilingual reasoning capability across diverse languages and cultural contexts remains limited. Existing multilingual reasoning benchmarks are typically constructed by translating existing English reasoning benchmarks, biasing these benchmarks towards reasoning problems with context in English language/cultures. In this work, we introduce the Multilingual Native Reasoning Challenge (MultiNRC), a benchmark designed to assess LLMs on more than 1,000 native,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.17476","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-23T12:56:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ea339b5d64eae10c29fd5241184afa490f7c7999df88e05aee12d8d51c7c054f","abstract_canon_sha256":"517961789d5dbf254afdc6b0cf48de34340fe764082bf719dbc10855f6b123dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:08.550222Z","signature_b64":"0n+fz5hf8NOD+eMr0wmtQuj0zoRw7Ma8LemHwI5yy8zEJ3gHzqeXDBG+wBU/clAlLi3DvuUECLtsSHcd3f1NCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4f006ac21d69dfafb7b39470bf94b780e66950758d784641ab3c7bb0caa3565","last_reissued_at":"2026-07-05T11:42:08.549725Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:08.549725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MultiNRC: A Challenging and Native Multilingual Reasoning Evaluation Benchmark for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander R. Fabbri, Bing Liu, Chen Xing, Dean Lee, Diego Mares, Ernesto Hernandez, Jorge Flores, Meher Mankikar","submitted_at":"2025-07-23T12:56:31Z","abstract_excerpt":"Although recent Large Language Models (LLMs) have shown rapid improvement on reasoning benchmarks in English, the evaluation of such LLMs' multilingual reasoning capability across diverse languages and cultural contexts remains limited. Existing multilingual reasoning benchmarks are typically constructed by translating existing English reasoning benchmarks, biasing these benchmarks towards reasoning problems with context in English language/cultures. In this work, we introduce the Multilingual Native Reasoning Challenge (MultiNRC), a benchmark designed to assess LLMs on more than 1,000 native,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.17476","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.17476/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.17476","created_at":"2026-07-05T11:42:08.549794+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.17476v1","created_at":"2026-07-05T11:42:08.549794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.17476","created_at":"2026-07-05T11:42:08.549794+00:00"},{"alias_kind":"pith_short_12","alias_value":"2TYANLBB22O7","created_at":"2026-07-05T11:42:08.549794+00:00"},{"alias_kind":"pith_short_16","alias_value":"2TYANLBB22O7V633","created_at":"2026-07-05T11:42:08.549794+00:00"},{"alias_kind":"pith_short_8","alias_value":"2TYANLBB","created_at":"2026-07-05T11:42:08.549794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01879","citing_title":"CultureForest: Understanding and Evaluating Cultural Norm Grounded Reasoning in LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29630","citing_title":"SFBench: The SciFy Scientific Feasibility Benchmark","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26293","citing_title":"CroCo: Cross-Lingual Contrastive Preference Tuning on Self-Generations","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19262","citing_title":"CulturALL: Benchmarking Multilingual and Multicultural Competence of LLMs on Grounded Tasks","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA","json":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA.json","graph_json":"https://pith.science/api/pith-number/2TYANLBB22O7V633HFDQX6KLPA/graph.json","events_json":"https://pith.science/api/pith-number/2TYANLBB22O7V633HFDQX6KLPA/events.json","paper":"https://pith.science/paper/2TYANLBB"},"agent_actions":{"view_html":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA","download_json":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA.json","view_paper":"https://pith.science/paper/2TYANLBB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.17476&json=true","fetch_graph":"https://pith.science/api/pith-number/2TYANLBB22O7V633HFDQX6KLPA/graph.json","fetch_events":"https://pith.science/api/pith-number/2TYANLBB22O7V633HFDQX6KLPA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA/action/storage_attestation","attest_author":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA/action/author_attestation","sign_citation":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA/action/citation_signature","submit_replication":"https://pith.science/pith/2TYANLBB22O7V633HFDQX6KLPA/action/replication_record"}},"created_at":"2026-07-05T11:42:08.549794+00:00","updated_at":"2026-07-05T11:42:08.549794+00:00"}