{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3LY4CFZ3LE2QPY7SF3S37YN5DM","short_pith_number":"pith:3LY4CFZ3","schema_version":"1.0","canonical_sha256":"daf1c1173b593507e3f22ee5bfe1bd1b26bf41e47e42afec4dec41812324ce46","source":{"kind":"arxiv","id":"2508.04280","version":1},"attestation_state":"computed","paper":{"title":"Enhancing Vision-Language Model Training with Reinforcement Learning in Synthetic Worlds for Real-World Success","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniil Gavrilov, George Bredis, Ruslan Rakhimov, Stanislav Dereka, Viacheslav Sinii","submitted_at":"2025-08-06T10:08:48Z","abstract_excerpt":"Interactive multimodal agents must convert raw visual observations into coherent sequences of language-conditioned actions -- a capability that current vision-language models (VLMs) still lack. Earlier reinforcement-learning (RL) efforts could, in principle, endow VLMs with such skills, but they have seldom tested whether the learned behaviours generalize beyond their training simulators, and they depend either on brittle hyperparameter tuning or on dense-reward environments with low state variability. We introduce Vision-Language Decoupled Actor-Critic (VL-DAC), a lightweight, hyperparameter-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.04280","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-06T10:08:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7e7a7b619fdb4f7a1d5721b1317c116db1b824df32478f3fb37dff5d02df9db5","abstract_canon_sha256":"7fee53e915ebcb804fecb6f0aeccc615f8d0e9205a64928c165d6beace627091"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:24.171171Z","signature_b64":"omXboZNejJK8zlgc2JpOo93E5qmKAc5zeMEYa+sBY5tjvqWZGwrnZRYEwH4S/M4uAsUsNHjj+qQSesWH5/tCCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"daf1c1173b593507e3f22ee5bfe1bd1b26bf41e47e42afec4dec41812324ce46","last_reissued_at":"2026-07-05T11:49:24.170678Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:24.170678Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Vision-Language Model Training with Reinforcement Learning in Synthetic Worlds for Real-World Success","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniil Gavrilov, George Bredis, Ruslan Rakhimov, Stanislav Dereka, Viacheslav Sinii","submitted_at":"2025-08-06T10:08:48Z","abstract_excerpt":"Interactive multimodal agents must convert raw visual observations into coherent sequences of language-conditioned actions -- a capability that current vision-language models (VLMs) still lack. Earlier reinforcement-learning (RL) efforts could, in principle, endow VLMs with such skills, but they have seldom tested whether the learned behaviours generalize beyond their training simulators, and they depend either on brittle hyperparameter tuning or on dense-reward environments with low state variability. We introduce Vision-Language Decoupled Actor-Critic (VL-DAC), a lightweight, hyperparameter-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.04280","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.04280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.04280","created_at":"2026-07-05T11:49:24.170738+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.04280v1","created_at":"2026-07-05T11:49:24.170738+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.04280","created_at":"2026-07-05T11:49:24.170738+00:00"},{"alias_kind":"pith_short_12","alias_value":"3LY4CFZ3LE2Q","created_at":"2026-07-05T11:49:24.170738+00:00"},{"alias_kind":"pith_short_16","alias_value":"3LY4CFZ3LE2QPY7S","created_at":"2026-07-05T11:49:24.170738+00:00"},{"alias_kind":"pith_short_8","alias_value":"3LY4CFZ3","created_at":"2026-07-05T11:49:24.170738+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01897","citing_title":"Rank-Then-Act: Reward-Free Control from Frame-Order Progress","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09965","citing_title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09965","citing_title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07774","citing_title":"RoboAgent: Chaining Basic Capabilities for Embodied Task Planning","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM","json":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM.json","graph_json":"https://pith.science/api/pith-number/3LY4CFZ3LE2QPY7SF3S37YN5DM/graph.json","events_json":"https://pith.science/api/pith-number/3LY4CFZ3LE2QPY7SF3S37YN5DM/events.json","paper":"https://pith.science/paper/3LY4CFZ3"},"agent_actions":{"view_html":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM","download_json":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM.json","view_paper":"https://pith.science/paper/3LY4CFZ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.04280&json=true","fetch_graph":"https://pith.science/api/pith-number/3LY4CFZ3LE2QPY7SF3S37YN5DM/graph.json","fetch_events":"https://pith.science/api/pith-number/3LY4CFZ3LE2QPY7SF3S37YN5DM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM/action/storage_attestation","attest_author":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM/action/author_attestation","sign_citation":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM/action/citation_signature","submit_replication":"https://pith.science/pith/3LY4CFZ3LE2QPY7SF3S37YN5DM/action/replication_record"}},"created_at":"2026-07-05T11:49:24.170738+00:00","updated_at":"2026-07-05T11:49:24.170738+00:00"}