{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EPSGKYC2HCDHFZTEIZDSMCIVR6","short_pith_number":"pith:EPSGKYC2","schema_version":"1.0","canonical_sha256":"23e465605a388672e66446472609158f9e70334ad032f982bb0f864367de8ffb","source":{"kind":"arxiv","id":"2404.09486","version":2},"attestation_state":"computed","paper":{"title":"MMCode: Benchmarking Multimodal Large Language Models for Code Generation with Visually Rich Programming Problems","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV","cs.SE"],"primary_cat":"cs.CL","authors_text":"Jing Ma, Kaixin Li, Qisheng Hu, Yuchen Tian, Zhiyong Huang, Ziyang Luo","submitted_at":"2024-04-15T06:15:46Z","abstract_excerpt":"Programming often involves converting detailed and complex specifications into code, a process during which developers typically utilize visual aids to more effectively convey concepts. While recent developments in Large Multimodal Models have demonstrated remarkable abilities in visual reasoning and mathematical tasks, there is little work on investigating whether these models can effectively interpret visual elements for code generation. To this end, we present MMCode, the first multi-modal coding dataset for evaluating algorithmic problem-solving skills in visually rich contexts. MMCode con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.09486","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-15T06:15:46Z","cross_cats_sorted":["cs.CV","cs.SE"],"title_canon_sha256":"1e2e445cb01631291aeaecff8e48baa1fbffc2de856f55ab09e48813b471de2c","abstract_canon_sha256":"84b1372c7e0002e75d857b249df0d743c70a12ca868804ec7b119ebcc5ec0907"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:04.239881Z","signature_b64":"MrPeao77DwjJ7HPIJKZfrgGdys+ZtBhRIkSMNXQyUM9F9IuuY24p5fvCx6K8cbO0yfX1gxqoKss1+kcLLosrBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23e465605a388672e66446472609158f9e70334ad032f982bb0f864367de8ffb","last_reissued_at":"2026-07-05T09:12:04.239285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:04.239285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMCode: Benchmarking Multimodal Large Language Models for Code Generation with Visually Rich Programming Problems","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV","cs.SE"],"primary_cat":"cs.CL","authors_text":"Jing Ma, Kaixin Li, Qisheng Hu, Yuchen Tian, Zhiyong Huang, Ziyang Luo","submitted_at":"2024-04-15T06:15:46Z","abstract_excerpt":"Programming often involves converting detailed and complex specifications into code, a process during which developers typically utilize visual aids to more effectively convey concepts. While recent developments in Large Multimodal Models have demonstrated remarkable abilities in visual reasoning and mathematical tasks, there is little work on investigating whether these models can effectively interpret visual elements for code generation. To this end, we present MMCode, the first multi-modal coding dataset for evaluating algorithmic problem-solving skills in visually rich contexts. MMCode con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.09486","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.09486/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.09486","created_at":"2026-07-05T09:12:04.239377+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.09486v2","created_at":"2026-07-05T09:12:04.239377+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.09486","created_at":"2026-07-05T09:12:04.239377+00:00"},{"alias_kind":"pith_short_12","alias_value":"EPSGKYC2HCDH","created_at":"2026-07-05T09:12:04.239377+00:00"},{"alias_kind":"pith_short_16","alias_value":"EPSGKYC2HCDHFZTE","created_at":"2026-07-05T09:12:04.239377+00:00"},{"alias_kind":"pith_short_8","alias_value":"EPSGKYC2","created_at":"2026-07-05T09:12:04.239377+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02387","citing_title":"VS-Bench: Evaluating VLMs for Strategic Abilities in Multi-Agent Environments","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6","json":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6.json","graph_json":"https://pith.science/api/pith-number/EPSGKYC2HCDHFZTEIZDSMCIVR6/graph.json","events_json":"https://pith.science/api/pith-number/EPSGKYC2HCDHFZTEIZDSMCIVR6/events.json","paper":"https://pith.science/paper/EPSGKYC2"},"agent_actions":{"view_html":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6","download_json":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6.json","view_paper":"https://pith.science/paper/EPSGKYC2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.09486&json=true","fetch_graph":"https://pith.science/api/pith-number/EPSGKYC2HCDHFZTEIZDSMCIVR6/graph.json","fetch_events":"https://pith.science/api/pith-number/EPSGKYC2HCDHFZTEIZDSMCIVR6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6/action/storage_attestation","attest_author":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6/action/author_attestation","sign_citation":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6/action/citation_signature","submit_replication":"https://pith.science/pith/EPSGKYC2HCDHFZTEIZDSMCIVR6/action/replication_record"}},"created_at":"2026-07-05T09:12:04.239377+00:00","updated_at":"2026-07-05T09:12:04.239377+00:00"}