{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:DSGNU7YX65PMLZJERMR5Q5POWY","short_pith_number":"pith:DSGNU7YX","canonical_record":{"source":{"id":"2506.07202","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-08T15:52:38Z","cross_cats_sorted":[],"title_canon_sha256":"ec291dc7d6e7332f0117f3d649940c934580ec29a728345168665c0fddb9d0c4","abstract_canon_sha256":"a1795081b0f2c1f21ff98ac290c4ed50daaa5f25d9a994c43ea393e78fe6ca13"},"schema_version":"1.0"},"canonical_sha256":"1c8cda7f17f75ec5e5248b23d875eeb62dcbc4868ea172e8604d380577905eb1","source":{"kind":"arxiv","id":"2506.07202","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.07202","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"arxiv_version","alias_value":"2506.07202v1","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07202","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"pith_short_12","alias_value":"DSGNU7YX65PM","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"pith_short_16","alias_value":"DSGNU7YX65PMLZJE","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"pith_short_8","alias_value":"DSGNU7YX","created_at":"2026-07-05T11:18:03Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:DSGNU7YX65PMLZJERMR5Q5POWY","target":"record","payload":{"canonical_record":{"source":{"id":"2506.07202","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-08T15:52:38Z","cross_cats_sorted":[],"title_canon_sha256":"ec291dc7d6e7332f0117f3d649940c934580ec29a728345168665c0fddb9d0c4","abstract_canon_sha256":"a1795081b0f2c1f21ff98ac290c4ed50daaa5f25d9a994c43ea393e78fe6ca13"},"schema_version":"1.0"},"canonical_sha256":"1c8cda7f17f75ec5e5248b23d875eeb62dcbc4868ea172e8604d380577905eb1","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:03.433769Z","signature_b64":"bFD3O28sOZ5JwVkjAzcxungnuXxx2Ce6waq3FPGcneEcASHGB8tE1mouuBWDc1O2XtqeLt70WF9yavsN9+uXBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c8cda7f17f75ec5e5248b23d875eeb62dcbc4868ea172e8604d380577905eb1","last_reissued_at":"2026-07-05T11:18:03.433290Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:03.433290Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.07202","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:18:03Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JzFxjg8XVia9kL0MP6yFgVspYj+DQZ/C/O7uWeH8y+/nMWDUCZiq7eqWpE29BGnhIZ/Fx8sT1qivRkd/iFuIDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T18:53:06.038426Z"},"content_sha256":"6f94b43222714c2c37bf98e3bf726e39b7c15d0a3daeb20fe734880d0adcd09d","schema_version":"1.0","event_id":"sha256:6f94b43222714c2c37bf98e3bf726e39b7c15d0a3daeb20fe734880d0adcd09d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:DSGNU7YX65PMLZJERMR5Q5POWY","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reasoning Multimodal Large Language Model: Data Contamination and Dynamic Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ming Liu, Wensheng Zhang","submitted_at":"2025-06-08T15:52:38Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) show impressive vision-language benchmark performance, yet growing concerns about data contamination (test set exposure during training) risk masking true generalization. This concern extends to reasoning MLLMs, often fine-tuned via reinforcement learning from potentially contaminated base models. We propose a novel dynamic evaluation framework to rigorously assess MLLM generalization, moving beyond static benchmarks. Instead of perturbing inputs, we perturb the task itself. Using the same visual input, models are evaluated across a family of tasks (e.g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07202","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07202/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:18:03Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dPNt5hgh/wYfkF3DJ14D+FL6bp06iBGYievN9f/wxaPVVheA3YQieFcFdUmVCKg9VIYU2ZpZvdV0IGWx8tMIDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T18:53:06.038982Z"},"content_sha256":"893a701e8a9405f50e5ee190c8f7a392907cdb3e6499c959eddd29d88c70b870","schema_version":"1.0","event_id":"sha256:893a701e8a9405f50e5ee190c8f7a392907cdb3e6499c959eddd29d88c70b870"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/bundle.json","state_url":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/DSGNU7YX65PMLZJERMR5Q5POWY/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T18:53:06Z","links":{"resolver":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY","bundle":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/bundle.json","state":"https://pith.science/pith/DSGNU7YX65PMLZJERMR5Q5POWY/state.json","well_known_bundle":"https://pith.science/.well-known/pith/DSGNU7YX65PMLZJERMR5Q5POWY/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:DSGNU7YX65PMLZJERMR5Q5POWY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a1795081b0f2c1f21ff98ac290c4ed50daaa5f25d9a994c43ea393e78fe6ca13","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-08T15:52:38Z","title_canon_sha256":"ec291dc7d6e7332f0117f3d649940c934580ec29a728345168665c0fddb9d0c4"},"schema_version":"1.0","source":{"id":"2506.07202","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.07202","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"arxiv_version","alias_value":"2506.07202v1","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07202","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"pith_short_12","alias_value":"DSGNU7YX65PM","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"pith_short_16","alias_value":"DSGNU7YX65PMLZJE","created_at":"2026-07-05T11:18:03Z"},{"alias_kind":"pith_short_8","alias_value":"DSGNU7YX","created_at":"2026-07-05T11:18:03Z"}],"graph_snapshots":[{"event_id":"sha256:893a701e8a9405f50e5ee190c8f7a392907cdb3e6499c959eddd29d88c70b870","target":"graph","created_at":"2026-07-05T11:18:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.07202/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Multimodal Large Language Models (MLLMs) show impressive vision-language benchmark performance, yet growing concerns about data contamination (test set exposure during training) risk masking true generalization. This concern extends to reasoning MLLMs, often fine-tuned via reinforcement learning from potentially contaminated base models. We propose a novel dynamic evaluation framework to rigorously assess MLLM generalization, moving beyond static benchmarks. Instead of perturbing inputs, we perturb the task itself. Using the same visual input, models are evaluated across a family of tasks (e.g","authors_text":"Ming Liu, Wensheng Zhang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-08T15:52:38Z","title":"Reasoning Multimodal Large Language Model: Data Contamination and Dynamic Evaluation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07202","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6f94b43222714c2c37bf98e3bf726e39b7c15d0a3daeb20fe734880d0adcd09d","target":"record","created_at":"2026-07-05T11:18:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a1795081b0f2c1f21ff98ac290c4ed50daaa5f25d9a994c43ea393e78fe6ca13","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-08T15:52:38Z","title_canon_sha256":"ec291dc7d6e7332f0117f3d649940c934580ec29a728345168665c0fddb9d0c4"},"schema_version":"1.0","source":{"id":"2506.07202","kind":"arxiv","version":1}},"canonical_sha256":"1c8cda7f17f75ec5e5248b23d875eeb62dcbc4868ea172e8604d380577905eb1","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1c8cda7f17f75ec5e5248b23d875eeb62dcbc4868ea172e8604d380577905eb1","first_computed_at":"2026-07-05T11:18:03.433290Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:18:03.433290Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"bFD3O28sOZ5JwVkjAzcxungnuXxx2Ce6waq3FPGcneEcASHGB8tE1mouuBWDc1O2XtqeLt70WF9yavsN9+uXBg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:18:03.433769Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.07202","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6f94b43222714c2c37bf98e3bf726e39b7c15d0a3daeb20fe734880d0adcd09d","sha256:893a701e8a9405f50e5ee190c8f7a392907cdb3e6499c959eddd29d88c70b870"],"state_sha256":"5e7893ad207c8b1944f3ca1cfae6720d596545a34ac31bf38ecb92ab66fe6fde"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"r4ixLHv6b05ACWpL7mxbOt9GUa4vP5RLWD7OxlB9oqbrd2AMOTmxcjqmlXLIG35uWSNUqb/N48XpWIU6OZVOCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T18:53:06.043455Z","bundle_sha256":"8c6692404a5729849052b397c85f4dc1db066c998367f4a3adf2ed01cdcfc4be"}}