{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RXOR3DPPUUNM4WHM744MNG54WG","short_pith_number":"pith:RXOR3DPP","schema_version":"1.0","canonical_sha256":"8ddd1d8defa51ace58ecff38c69bbcb1b795480e7ca8e1e3b9a85c2341525d50","source":{"kind":"arxiv","id":"2505.12432","version":1},"attestation_state":"computed","paper":{"title":"Observe-R1: Unlocking Reasoning Abilities of MLLMs with Dynamic Progressive Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Minjie Hong, Tao Jin, Zirun Guo","submitted_at":"2025-05-18T14:08:03Z","abstract_excerpt":"Reinforcement Learning (RL) has shown promise in improving the reasoning abilities of Large Language Models (LLMs). However, the specific challenges of adapting RL to multimodal data and formats remain relatively unexplored. In this work, we present Observe-R1, a novel framework aimed at enhancing the reasoning capabilities of multimodal large language models (MLLMs). We draw inspirations from human learning progression--from simple to complex and easy to difficult, and propose a gradual learning paradigm for MLLMs. To this end, we construct the NeuraLadder dataset, which is organized and samp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12432","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-18T14:08:03Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"e4ae891e5c308e5a86bb0275857d534a17bef08b8f1bcc70cd9303d18aa2a390","abstract_canon_sha256":"6908376a37abc949cb6e5d3fd22b4fb3a5de11d3001d0e3185ca0cd48e6c44da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:04.119625Z","signature_b64":"Cok7v+jJbuJ/sBa4lGXL4ZP3WdQuZjWgPDaIbr04fbueKIEuU3pnsrcsSdRdfXpqIqN/bVm36BUqe682fpX+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ddd1d8defa51ace58ecff38c69bbcb1b795480e7ca8e1e3b9a85c2341525d50","last_reissued_at":"2026-07-05T11:05:04.119175Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:04.119175Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Observe-R1: Unlocking Reasoning Abilities of MLLMs with Dynamic Progressive Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Minjie Hong, Tao Jin, Zirun Guo","submitted_at":"2025-05-18T14:08:03Z","abstract_excerpt":"Reinforcement Learning (RL) has shown promise in improving the reasoning abilities of Large Language Models (LLMs). However, the specific challenges of adapting RL to multimodal data and formats remain relatively unexplored. In this work, we present Observe-R1, a novel framework aimed at enhancing the reasoning capabilities of multimodal large language models (MLLMs). We draw inspirations from human learning progression--from simple to complex and easy to difficult, and propose a gradual learning paradigm for MLLMs. To this end, we construct the NeuraLadder dataset, which is organized and samp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12432","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12432/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12432","created_at":"2026-07-05T11:05:04.119243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12432v1","created_at":"2026-07-05T11:05:04.119243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12432","created_at":"2026-07-05T11:05:04.119243+00:00"},{"alias_kind":"pith_short_12","alias_value":"RXOR3DPPUUNM","created_at":"2026-07-05T11:05:04.119243+00:00"},{"alias_kind":"pith_short_16","alias_value":"RXOR3DPPUUNM4WHM","created_at":"2026-07-05T11:05:04.119243+00:00"},{"alias_kind":"pith_short_8","alias_value":"RXOR3DPP","created_at":"2026-07-05T11:05:04.119243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":241,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG","json":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG.json","graph_json":"https://pith.science/api/pith-number/RXOR3DPPUUNM4WHM744MNG54WG/graph.json","events_json":"https://pith.science/api/pith-number/RXOR3DPPUUNM4WHM744MNG54WG/events.json","paper":"https://pith.science/paper/RXOR3DPP"},"agent_actions":{"view_html":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG","download_json":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG.json","view_paper":"https://pith.science/paper/RXOR3DPP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12432&json=true","fetch_graph":"https://pith.science/api/pith-number/RXOR3DPPUUNM4WHM744MNG54WG/graph.json","fetch_events":"https://pith.science/api/pith-number/RXOR3DPPUUNM4WHM744MNG54WG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG/action/storage_attestation","attest_author":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG/action/author_attestation","sign_citation":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG/action/citation_signature","submit_replication":"https://pith.science/pith/RXOR3DPPUUNM4WHM744MNG54WG/action/replication_record"}},"created_at":"2026-07-05T11:05:04.119243+00:00","updated_at":"2026-07-05T11:05:04.119243+00:00"}