{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YYSA6VKMODSOKVPA227ZJWYNWW","short_pith_number":"pith:YYSA6VKM","schema_version":"1.0","canonical_sha256":"c6240f554c70e4e555e0d6bf94db0db590f987429b67a73ef49f21ec6e350b86","source":{"kind":"arxiv","id":"2508.10137","version":1},"attestation_state":"computed","paper":{"title":"mSCoRe: a $M$ultilingual and Scalable Benchmark for $S$kill-based $Co$mmonsense $Re$asoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Franck Dernoncourt, Nghia Trung Ngo, Thien Huu Nguyen","submitted_at":"2025-08-13T18:59:02Z","abstract_excerpt":"Recent advancements in reasoning-reinforced Large Language Models (LLMs) have shown remarkable capabilities in complex reasoning tasks. However, the mechanism underlying their utilization of different human reasoning skills remains poorly investigated, especially for multilingual commonsense reasoning that involves everyday knowledge across different languages and cultures. To address this gap, we propose a \\textbf{M}ultilingual and Scalable Benchmark for \\textbf{S}kill-based \\textbf{Co}mmonsense \\textbf{Re}asoning (\\textbf{mSCoRe}). Our benchmark incorporates three key components that are des"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.10137","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-13T18:59:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"70845cffda586f88b14d3786c7b8e6b3685aa4a27571b1f94b0e4f5126ff967e","abstract_canon_sha256":"7ba636aface47004bfa153a5248ce067d4c496247d9d5f839a6987d109582d4e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:32.325523Z","signature_b64":"sk/H4BlpzQWfTOwGWSpehUlxYYZnl9p5/+aPhYQkCp1ANs8IM5H+Z25GKUPxsXg983Eh+6L9nIWmL2DylEcPDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c6240f554c70e4e555e0d6bf94db0db590f987429b67a73ef49f21ec6e350b86","last_reissued_at":"2026-07-05T11:53:32.324951Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:32.324951Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"mSCoRe: a $M$ultilingual and Scalable Benchmark for $S$kill-based $Co$mmonsense $Re$asoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Franck Dernoncourt, Nghia Trung Ngo, Thien Huu Nguyen","submitted_at":"2025-08-13T18:59:02Z","abstract_excerpt":"Recent advancements in reasoning-reinforced Large Language Models (LLMs) have shown remarkable capabilities in complex reasoning tasks. However, the mechanism underlying their utilization of different human reasoning skills remains poorly investigated, especially for multilingual commonsense reasoning that involves everyday knowledge across different languages and cultures. To address this gap, we propose a \\textbf{M}ultilingual and Scalable Benchmark for \\textbf{S}kill-based \\textbf{Co}mmonsense \\textbf{Re}asoning (\\textbf{mSCoRe}). Our benchmark incorporates three key components that are des"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.10137","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.10137/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.10137","created_at":"2026-07-05T11:53:32.325019+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.10137v1","created_at":"2026-07-05T11:53:32.325019+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.10137","created_at":"2026-07-05T11:53:32.325019+00:00"},{"alias_kind":"pith_short_12","alias_value":"YYSA6VKMODSO","created_at":"2026-07-05T11:53:32.325019+00:00"},{"alias_kind":"pith_short_16","alias_value":"YYSA6VKMODSOKVPA","created_at":"2026-07-05T11:53:32.325019+00:00"},{"alias_kind":"pith_short_8","alias_value":"YYSA6VKM","created_at":"2026-07-05T11:53:32.325019+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":178,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW","json":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW.json","graph_json":"https://pith.science/api/pith-number/YYSA6VKMODSOKVPA227ZJWYNWW/graph.json","events_json":"https://pith.science/api/pith-number/YYSA6VKMODSOKVPA227ZJWYNWW/events.json","paper":"https://pith.science/paper/YYSA6VKM"},"agent_actions":{"view_html":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW","download_json":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW.json","view_paper":"https://pith.science/paper/YYSA6VKM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.10137&json=true","fetch_graph":"https://pith.science/api/pith-number/YYSA6VKMODSOKVPA227ZJWYNWW/graph.json","fetch_events":"https://pith.science/api/pith-number/YYSA6VKMODSOKVPA227ZJWYNWW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW/action/storage_attestation","attest_author":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW/action/author_attestation","sign_citation":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW/action/citation_signature","submit_replication":"https://pith.science/pith/YYSA6VKMODSOKVPA227ZJWYNWW/action/replication_record"}},"created_at":"2026-07-05T11:53:32.325019+00:00","updated_at":"2026-07-05T11:53:32.325019+00:00"}