{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3R35MEDNPFPS6GNBNOEQWTUNES","short_pith_number":"pith:3R35MEDN","schema_version":"1.0","canonical_sha256":"dc77d6106d795f2f19a16b890b4e8d24b91a6d1e86525e22ca0e28e200cf6025","source":{"kind":"arxiv","id":"2507.11882","version":1},"attestation_state":"computed","paper":{"title":"Marco-Bench-MIF: On Multilingual Instruction-Following Capability of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bo Zeng, Chenyang Lyu, Chenyu Zhu, Jiahui Geng, Kaifu Zhang, Longyue Wang, Minghao Wu, Mingyan Zeng, Qing Li, Ruizhe Li, Sinuo Liu, Tianqi Shi, Weihua Luo, Xuanfan Ni, Yefeng Liu, Yu Tong, Yu Zhao","submitted_at":"2025-07-16T03:49:41Z","abstract_excerpt":"Instruction-following capability has become a major ability to be evaluated for Large Language Models (LLMs). However, existing datasets, such as IFEval, are either predominantly monolingual and centered on English or simply machine translated to other languages, limiting their applicability in multilingual contexts. In this paper, we present an carefully-curated extension of IFEval to a localized multilingual version named Marco-Bench-MIF, covering 30 languages with varying levels of localization. Our benchmark addresses linguistic constraints (e.g., modifying capitalization requirements for "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.11882","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-16T03:49:41Z","cross_cats_sorted":[],"title_canon_sha256":"1e6773bd533cd9d8af61f5067b7a729b74990ce2fca786120edf27cde10355d1","abstract_canon_sha256":"b05f68fc542ed17573c8fe0f8b64a786c44228ed6ad846633b64409194958307"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:37:52.712285Z","signature_b64":"2O/fZPb4rDtK/RFoM3e5IhAdPkAB7sCnWxmsdLaWlQVwbrkbzPfeviqdGwZfnl5rDugCQ/cVUTrbA9AlNDQEDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc77d6106d795f2f19a16b890b4e8d24b91a6d1e86525e22ca0e28e200cf6025","last_reissued_at":"2026-07-05T11:37:52.711456Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:37:52.711456Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Marco-Bench-MIF: On Multilingual Instruction-Following Capability of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bo Zeng, Chenyang Lyu, Chenyu Zhu, Jiahui Geng, Kaifu Zhang, Longyue Wang, Minghao Wu, Mingyan Zeng, Qing Li, Ruizhe Li, Sinuo Liu, Tianqi Shi, Weihua Luo, Xuanfan Ni, Yefeng Liu, Yu Tong, Yu Zhao","submitted_at":"2025-07-16T03:49:41Z","abstract_excerpt":"Instruction-following capability has become a major ability to be evaluated for Large Language Models (LLMs). However, existing datasets, such as IFEval, are either predominantly monolingual and centered on English or simply machine translated to other languages, limiting their applicability in multilingual contexts. In this paper, we present an carefully-curated extension of IFEval to a localized multilingual version named Marco-Bench-MIF, covering 30 languages with varying levels of localization. Our benchmark addresses linguistic constraints (e.g., modifying capitalization requirements for "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.11882","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.11882/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.11882","created_at":"2026-07-05T11:37:52.711539+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.11882v1","created_at":"2026-07-05T11:37:52.711539+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.11882","created_at":"2026-07-05T11:37:52.711539+00:00"},{"alias_kind":"pith_short_12","alias_value":"3R35MEDNPFPS","created_at":"2026-07-05T11:37:52.711539+00:00"},{"alias_kind":"pith_short_16","alias_value":"3R35MEDNPFPS6GNB","created_at":"2026-07-05T11:37:52.711539+00:00"},{"alias_kind":"pith_short_8","alias_value":"3R35MEDN","created_at":"2026-07-05T11:37:52.711539+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES","json":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES.json","graph_json":"https://pith.science/api/pith-number/3R35MEDNPFPS6GNBNOEQWTUNES/graph.json","events_json":"https://pith.science/api/pith-number/3R35MEDNPFPS6GNBNOEQWTUNES/events.json","paper":"https://pith.science/paper/3R35MEDN"},"agent_actions":{"view_html":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES","download_json":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES.json","view_paper":"https://pith.science/paper/3R35MEDN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.11882&json=true","fetch_graph":"https://pith.science/api/pith-number/3R35MEDNPFPS6GNBNOEQWTUNES/graph.json","fetch_events":"https://pith.science/api/pith-number/3R35MEDNPFPS6GNBNOEQWTUNES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES/action/storage_attestation","attest_author":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES/action/author_attestation","sign_citation":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES/action/citation_signature","submit_replication":"https://pith.science/pith/3R35MEDNPFPS6GNBNOEQWTUNES/action/replication_record"}},"created_at":"2026-07-05T11:37:52.711539+00:00","updated_at":"2026-07-05T11:37:52.711539+00:00"}