{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:U3ZZYYWM5AADSE4T7O2L4H4VEP","short_pith_number":"pith:U3ZZYYWM","schema_version":"1.0","canonical_sha256":"a6f39c62cce800391393fbb4be1f9523d7bb79301d1d4d1139ff20a2f3f2dc78","source":{"kind":"arxiv","id":"2607.13854","version":1},"attestation_state":"computed","paper":{"title":"SPyCE: Skill-Policy Co-evolution for Multimodal Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ru Zhang, Weijie Qiu","submitted_at":"2026-07-15T14:01:48Z","abstract_excerpt":"Multimodal agents that think with images iteratively manipulate visual evidence and invoke tools across many steps. Existing reinforcement learning methods reduce trajectories to scalar rewards, forcing the policy to discover reusable tool-use patterns from scratch on every new task; memory-based alternatives retain past experience, yet they rely on test-time retrieval, without updating the policy to absorb reusable patterns from that experience. Our key insight is that multimodal reasoning trajectories should be distilled into reusable skills that co-evolve with the policy during training, ra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.13854","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-15T14:01:48Z","cross_cats_sorted":[],"title_canon_sha256":"5159e4e72d13645432812734e93cdacf7328f43274335bae6930ac13370f10b5","abstract_canon_sha256":"61c64bd87c481b115e27fe52ee835443a29bd70c6d5820cff23083905725595a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-16T01:23:09.951722Z","signature_b64":"yjhUrBk/gHcwMG9bKkMZKNGsACjBJV8QNi9diCMN3BplL/qiI+do1A4O7PRz631jIGEPXxGQYZfl5BobSQTxDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6f39c62cce800391393fbb4be1f9523d7bb79301d1d4d1139ff20a2f3f2dc78","last_reissued_at":"2026-07-16T01:23:09.950707Z","signature_status":"signed_v1","first_computed_at":"2026-07-16T01:23:09.950707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SPyCE: Skill-Policy Co-evolution for Multimodal Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ru Zhang, Weijie Qiu","submitted_at":"2026-07-15T14:01:48Z","abstract_excerpt":"Multimodal agents that think with images iteratively manipulate visual evidence and invoke tools across many steps. Existing reinforcement learning methods reduce trajectories to scalar rewards, forcing the policy to discover reusable tool-use patterns from scratch on every new task; memory-based alternatives retain past experience, yet they rely on test-time retrieval, without updating the policy to absorb reusable patterns from that experience. Our key insight is that multimodal reasoning trajectories should be distilled into reusable skills that co-evolve with the policy during training, ra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.13854","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.13854/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.13854","created_at":"2026-07-16T01:23:09.951239+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.13854v1","created_at":"2026-07-16T01:23:09.951239+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.13854","created_at":"2026-07-16T01:23:09.951239+00:00"},{"alias_kind":"pith_short_12","alias_value":"U3ZZYYWM5AAD","created_at":"2026-07-16T01:23:09.951239+00:00"},{"alias_kind":"pith_short_16","alias_value":"U3ZZYYWM5AADSE4T","created_at":"2026-07-16T01:23:09.951239+00:00"},{"alias_kind":"pith_short_8","alias_value":"U3ZZYYWM","created_at":"2026-07-16T01:23:09.951239+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP","json":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP.json","graph_json":"https://pith.science/api/pith-number/U3ZZYYWM5AADSE4T7O2L4H4VEP/graph.json","events_json":"https://pith.science/api/pith-number/U3ZZYYWM5AADSE4T7O2L4H4VEP/events.json","paper":"https://pith.science/paper/U3ZZYYWM"},"agent_actions":{"view_html":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP","download_json":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP.json","view_paper":"https://pith.science/paper/U3ZZYYWM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.13854&json=true","fetch_graph":"https://pith.science/api/pith-number/U3ZZYYWM5AADSE4T7O2L4H4VEP/graph.json","fetch_events":"https://pith.science/api/pith-number/U3ZZYYWM5AADSE4T7O2L4H4VEP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP/action/storage_attestation","attest_author":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP/action/author_attestation","sign_citation":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP/action/citation_signature","submit_replication":"https://pith.science/pith/U3ZZYYWM5AADSE4T7O2L4H4VEP/action/replication_record"}},"created_at":"2026-07-16T01:23:09.951239+00:00","updated_at":"2026-07-16T01:23:09.951239+00:00"}