{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:JCG5EUW6XF6MYC53WONCPOM2IB","short_pith_number":"pith:JCG5EUW6","canonical_record":{"source":{"id":"2512.03438","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-03T04:42:47Z","cross_cats_sorted":[],"title_canon_sha256":"240e645706406640aaa672c3ed1394978de8bf16e4adc23dd9b3d2faa2ff0947","abstract_canon_sha256":"2ccd31160a2205f68c3b8a80a428e7b9d798b4500ac648cf9dfd1072b455b266"},"schema_version":"1.0"},"canonical_sha256":"488dd252deb97ccc0bbbb39a27b99a405e791cbce90fd585b68bd6c08fece5c8","source":{"kind":"arxiv","id":"2512.03438","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2512.03438","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"arxiv_version","alias_value":"2512.03438v3","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.03438","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"pith_short_12","alias_value":"JCG5EUW6XF6M","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"pith_short_16","alias_value":"JCG5EUW6XF6MYC53","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"pith_short_8","alias_value":"JCG5EUW6","created_at":"2026-08-03T01:12:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:JCG5EUW6XF6MYC53WONCPOM2IB","target":"record","payload":{"canonical_record":{"source":{"id":"2512.03438","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-03T04:42:47Z","cross_cats_sorted":[],"title_canon_sha256":"240e645706406640aaa672c3ed1394978de8bf16e4adc23dd9b3d2faa2ff0947","abstract_canon_sha256":"2ccd31160a2205f68c3b8a80a428e7b9d798b4500ac648cf9dfd1072b455b266"},"schema_version":"1.0"},"canonical_sha256":"488dd252deb97ccc0bbbb39a27b99a405e791cbce90fd585b68bd6c08fece5c8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-03T01:12:16.503038Z","signature_b64":"W/sgZPhjfmzpDruxy1JycueWnvF8XIJQJOWwvJw6Kg+KxCk13VSSSsUxdq6ZoC6o6vK11Tila45oKu4R9OEjCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"488dd252deb97ccc0bbbb39a27b99a405e791cbce90fd585b68bd6c08fece5c8","last_reissued_at":"2026-08-03T01:12:16.501505Z","signature_status":"signed_v1","first_computed_at":"2026-08-03T01:12:16.501505Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2512.03438","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-03T01:12:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ZucAAKIwj1KgjfjaOFaiC+NWCfLjtDfI6fejuNqMLgiYdQV+pmRuvtogBBdoH+STR+j0+cIjapCv5KCAZDrHAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T18:58:26.850013Z"},"content_sha256":"62927405a4c56e7f20735a2ac1a798c5b0a071f54df20e8257483e44522d9b95","schema_version":"1.0","event_id":"sha256:62927405a4c56e7f20735a2ac1a798c5b0a071f54df20e8257483e44522d9b95"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:JCG5EUW6XF6MYC53WONCPOM2IB","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"An adaptive verifier selects scoring functions during multimodal RL to achieve state-of-the-art results on agentic tasks.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andrea Tupini, Baolin Peng, Hao Cheng, Isar Meijer, Jianfeng Gao, Lars Liden, Lijuan Wang, Marc Pollefeys, Oier Mees, Qianhui Wu, Reuben Tan, Sheng Zhang, Theodore Zhao, XiaoDong Liu, Yong Jae Lee, Yu Gu, Yuncong Yang, Zhengyuan Yang","submitted_at":"2025-12-03T04:42:47Z","abstract_excerpt":"Agentic reasoning models trained with multimodal reinforcement learning (MMRL) have become increasingly capable, yet they are almost universally optimized using sparse, outcome-based rewards computed based on the final answers. Richer rewards computed from the reasoning tokens can improve learning significantly by providing more fine-grained guidance. However, it is challenging to compute more informative rewards in MMRL beyond those based on outcomes since different samples may require different scoring functions and teacher models may provide noisy reward signals too. In this paper, we intro"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"by leveraging our agentic verifier across both SFT data curation and RL training, our model achieves state-of-the-art results across multiple agentic tasks such as spatial reasoning, visual hallucination as well as robotics and embodied AI benchmarks.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That a pool of teacher-model-derived and rule-based scoring functions exists which, when adaptively selected by Argos, consistently provides more informative and less noisy rewards than outcome-based signals alone without introducing new biases or selection artifacts.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Argos is an agentic verifier that adaptively picks scoring functions to evaluate accuracy, localization, and reasoning quality, enabling stronger multimodal RL training for AI agents.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"An adaptive verifier selects scoring functions during multimodal RL to achieve state-of-the-art results on agentic tasks.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"b1674996100f4a7697f5c95f991dab70e8a5d5a597e70a734eba0469c02436d8"},"source":{"id":"2512.03438","kind":"arxiv","version":3},"verdict":{"id":"ee0ca933-755f-4d00-a5cc-c6bf8a1ff275","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-17T03:05:01.265138Z","strongest_claim":"by leveraging our agentic verifier across both SFT data curation and RL training, our model achieves state-of-the-art results across multiple agentic tasks such as spatial reasoning, visual hallucination as well as robotics and embodied AI benchmarks.","one_line_summary":"Argos is an agentic verifier that adaptively picks scoring functions to evaluate accuracy, localization, and reasoning quality, enabling stronger multimodal RL training for AI agents.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That a pool of teacher-model-derived and rule-based scoring functions exists which, when adaptively selected by Argos, consistently provides more informative and less noisy rewards than outcome-based signals alone without introducing new biases or selection artifacts.","pith_extraction_headline":"An adaptive verifier selects scoring functions during multimodal RL to achieve state-of-the-art results on agentic tasks."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.03438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"adc39d6ef36cfc08201323f9b397f3415116ddfbc3cae5b2a15109f46772af89"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"ee0ca933-755f-4d00-a5cc-c6bf8a1ff275"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-03T01:12:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"S1HYT6ZTk1DiuDpqZsUzHYd3aoCUAfMqCGoCKIZ1IWiVIAL/3WZG2NnvXXF3S5kvXamvKfdcTj4XZjElMuaTBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T18:58:26.850658Z"},"content_sha256":"b5d5dcd75eabf4fd4edbfe3d454ef504e8542534579d13f75875d3efffdb895d","schema_version":"1.0","event_id":"sha256:b5d5dcd75eabf4fd4edbfe3d454ef504e8542534579d13f75875d3efffdb895d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JCG5EUW6XF6MYC53WONCPOM2IB/bundle.json","state_url":"https://pith.science/pith/JCG5EUW6XF6MYC53WONCPOM2IB/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JCG5EUW6XF6MYC53WONCPOM2IB/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T18:58:26Z","links":{"resolver":"https://pith.science/pith/JCG5EUW6XF6MYC53WONCPOM2IB","bundle":"https://pith.science/pith/JCG5EUW6XF6MYC53WONCPOM2IB/bundle.json","state":"https://pith.science/pith/JCG5EUW6XF6MYC53WONCPOM2IB/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JCG5EUW6XF6MYC53WONCPOM2IB/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:JCG5EUW6XF6MYC53WONCPOM2IB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2ccd31160a2205f68c3b8a80a428e7b9d798b4500ac648cf9dfd1072b455b266","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-03T04:42:47Z","title_canon_sha256":"240e645706406640aaa672c3ed1394978de8bf16e4adc23dd9b3d2faa2ff0947"},"schema_version":"1.0","source":{"id":"2512.03438","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2512.03438","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"arxiv_version","alias_value":"2512.03438v3","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.03438","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"pith_short_12","alias_value":"JCG5EUW6XF6M","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"pith_short_16","alias_value":"JCG5EUW6XF6MYC53","created_at":"2026-08-03T01:12:16Z"},{"alias_kind":"pith_short_8","alias_value":"JCG5EUW6","created_at":"2026-08-03T01:12:16Z"}],"graph_snapshots":[{"event_id":"sha256:b5d5dcd75eabf4fd4edbfe3d454ef504e8542534579d13f75875d3efffdb895d","target":"graph","created_at":"2026-08-03T01:12:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"by leveraging our agentic verifier across both SFT data curation and RL training, our model achieves state-of-the-art results across multiple agentic tasks such as spatial reasoning, visual hallucination as well as robotics and embodied AI benchmarks."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That a pool of teacher-model-derived and rule-based scoring functions exists which, when adaptively selected by Argos, consistently provides more informative and less noisy rewards than outcome-based signals alone without introducing new biases or selection artifacts."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Argos is an agentic verifier that adaptively picks scoring functions to evaluate accuracy, localization, and reasoning quality, enabling stronger multimodal RL training for AI agents."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"An adaptive verifier selects scoring functions during multimodal RL to achieve state-of-the-art results on agentic tasks."}],"snapshot_sha256":"b1674996100f4a7697f5c95f991dab70e8a5d5a597e70a734eba0469c02436d8"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"adc39d6ef36cfc08201323f9b397f3415116ddfbc3cae5b2a15109f46772af89"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2512.03438/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Agentic reasoning models trained with multimodal reinforcement learning (MMRL) have become increasingly capable, yet they are almost universally optimized using sparse, outcome-based rewards computed based on the final answers. Richer rewards computed from the reasoning tokens can improve learning significantly by providing more fine-grained guidance. However, it is challenging to compute more informative rewards in MMRL beyond those based on outcomes since different samples may require different scoring functions and teacher models may provide noisy reward signals too. In this paper, we intro","authors_text":"Andrea Tupini, Baolin Peng, Hao Cheng, Isar Meijer, Jianfeng Gao, Lars Liden, Lijuan Wang, Marc Pollefeys, Oier Mees, Qianhui Wu, Reuben Tan, Sheng Zhang, Theodore Zhao, XiaoDong Liu, Yong Jae Lee, Yu Gu, Yuncong Yang, Zhengyuan Yang","cross_cats":[],"headline":"An adaptive verifier selects scoring functions during multimodal RL to achieve state-of-the-art results on agentic tasks.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-03T04:42:47Z","title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.03438","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-17T03:05:01.265138Z","id":"ee0ca933-755f-4d00-a5cc-c6bf8a1ff275","model_set":{"reader":"grok-4.3"},"one_line_summary":"Argos is an agentic verifier that adaptively picks scoring functions to evaluate accuracy, localization, and reasoning quality, enabling stronger multimodal RL training for AI agents.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"An adaptive verifier selects scoring functions during multimodal RL to achieve state-of-the-art results on agentic tasks.","strongest_claim":"by leveraging our agentic verifier across both SFT data curation and RL training, our model achieves state-of-the-art results across multiple agentic tasks such as spatial reasoning, visual hallucination as well as robotics and embodied AI benchmarks.","weakest_assumption":"That a pool of teacher-model-derived and rule-based scoring functions exists which, when adaptively selected by Argos, consistently provides more informative and less noisy rewards than outcome-based signals alone without introducing new biases or selection artifacts."}},"verdict_id":"ee0ca933-755f-4d00-a5cc-c6bf8a1ff275"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:62927405a4c56e7f20735a2ac1a798c5b0a071f54df20e8257483e44522d9b95","target":"record","created_at":"2026-08-03T01:12:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2ccd31160a2205f68c3b8a80a428e7b9d798b4500ac648cf9dfd1072b455b266","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-03T04:42:47Z","title_canon_sha256":"240e645706406640aaa672c3ed1394978de8bf16e4adc23dd9b3d2faa2ff0947"},"schema_version":"1.0","source":{"id":"2512.03438","kind":"arxiv","version":3}},"canonical_sha256":"488dd252deb97ccc0bbbb39a27b99a405e791cbce90fd585b68bd6c08fece5c8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"488dd252deb97ccc0bbbb39a27b99a405e791cbce90fd585b68bd6c08fece5c8","first_computed_at":"2026-08-03T01:12:16.501505Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-03T01:12:16.501505Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"W/sgZPhjfmzpDruxy1JycueWnvF8XIJQJOWwvJw6Kg+KxCk13VSSSsUxdq6ZoC6o6vK11Tila45oKu4R9OEjCQ==","signature_status":"signed_v1","signed_at":"2026-08-03T01:12:16.503038Z","signed_message":"canonical_sha256_bytes"},"source_id":"2512.03438","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:62927405a4c56e7f20735a2ac1a798c5b0a071f54df20e8257483e44522d9b95","sha256:b5d5dcd75eabf4fd4edbfe3d454ef504e8542534579d13f75875d3efffdb895d"],"state_sha256":"aeea6f8b2fbb8e8ad47247a8d8a5ae5b27d09d04c1d81adf99fd9550f028eb7d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"h5y7F7OCoAZxYLMwdSUoGCa3Gz4ACpTSj1A+7MEPozba+gYUzkSjOlUP6GCTOw/c4D8+xRcRi4Kwcv3mLq/EDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T18:58:26.855661Z","bundle_sha256":"ec0273a77ba17f26877bdecd661fbd911ebcd30c7ee88b8f50ac85e25f0a4560"}}