{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DSGNU7YX65PMLZJERMR5Q5POWY","short_pith_number":"pith:DSGNU7YX","schema_version":"1.0","canonical_sha256":"1c8cda7f17f75ec5e5248b23d875eeb62dcbc4868ea172e8604d380577905eb1","source":{"kind":"arxiv","id":"2506.07202","version":1},"attestation_state":"computed","paper":{"title":"Reasoning Multimodal Large Language Model: Data Contamination and Dynamic Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ming Liu, Wensheng Zhang","submitted_at":"2025-06-08T15:52:38Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) show impressive vision-language benchmark performance, yet growing concerns about data contamination (test set exposure during training) risk masking true generalization. This concern extends to reasoning MLLMs, often fine-tuned via reinforcement learning from potentially contaminated base models. We propose a novel dynamic evaluation framework to rigorously assess MLLM generalization, moving beyond static benchmarks. Instead of perturbing inputs, we perturb the task itself. Using the same visual input, models are evaluated across a family of tasks (e.g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07202","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-08T15:52:38Z","cross_cats_sorted":[],"title_canon_sha256":"ec291dc7d6e7332f0117f3d649940c934580ec29a728345168665c0fddb9d0c4","abstract_canon_sha256":"a1795081b0f2c1f21ff98ac290c4ed50daaa5f25d9a994c43ea393e78fe6ca13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:03.433769Z","signature_b64":"bFD3O28sOZ5JwVkjAzcxungnuXxx2Ce6waq3FPGcneEcASHGB8tE1mouuBWDc1O2XtqeLt70WF9yavsN9+uXBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c8cda7f17f75ec5e5248b23d875eeb62dcbc4868ea172e8604d380577905eb1","last_reissued_at":"2026-07-05T11:18:03.433290Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:03.433290Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning Multimodal Large Language Model: Data Contamination and Dynamic Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ming Liu, Wensheng Zhang","submitted_at":"2025-06-08T15:52:38Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) show impressive vision-language benchmark performance, yet growing concerns about data contamination (test set exposure during training) risk masking true generalization. This concern extends to reasoning MLLMs, often fine-tuned via reinforcement learning from potentially contaminated base models. We propose a novel dynamic evaluation framework to rigorously assess MLLM generalization, moving beyond static benchmarks. Instead of perturbing inputs, we perturb the task itself. Using the same visual input, models are evaluated across a family of tasks (e.g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07202","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07202/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07202","created_at":"2026-07-05T11:18:03.433356+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07202v1","created_at":"2026-07-05T11:18:03.433356+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07202","created_at":"2026-07-05T11:18:03.433356+00:00"},{"alias_kind":"pith_short_12","alias_value":"DSGNU7YX65PM","created_at":"2026-07-05T11:18:03.433356+00:00"},{"alias_kind":"pith_short_16","alias_value":"DSGNU7YX65PMLZJE","created_at":"2026-07-05T11:18:03.433356+00:00"},{"alias_kind":"pith_short_8","alias_value":"DSGNU7YX","created_at":"2026-07-05T11:18:03.433356+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29339","citing_title":"DMC-CF: Dynamic Multimodal CounterFactual QA benchmark for Causal Reasoning","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY","json":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY.json","graph_json":"https://pith.science/api/pith-number/DSGNU7YX65PMLZJERMR5Q5POWY/graph.json","events_json":"https://pith.science/api/pith-number/DSGNU7YX65PMLZJERMR5Q5POWY/events.json","paper":"https://pith.science/paper/DSGNU7YX"},"agent_actions":{"view_html":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY","download_json":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY.json","view_paper":"https://pith.science/paper/DSGNU7YX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07202&json=true","fetch_graph":"https://pith.science/api/pith-number/DSGNU7YX65PMLZJERMR5Q5POWY/graph.json","fetch_events":"https://pith.science/api/pith-number/DSGNU7YX65PMLZJERMR5Q5POWY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/action/storage_attestation","attest_author":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/action/author_attestation","sign_citation":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/action/citation_signature","submit_replication":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/action/replication_record"}},"created_at":"2026-07-05T11:18:03.433356+00:00","updated_at":"2026-07-05T11:18:03.433356+00:00"}