{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MJFJIVCFUMSPXMTQVPICWXCL5W","short_pith_number":"pith:MJFJIVCF","schema_version":"1.0","canonical_sha256":"624a945445a324fbb270abd02b5c4bedbdc7196e9cb9d563b99505c684deb77f","source":{"kind":"arxiv","id":"2506.22624","version":1},"attestation_state":"computed","paper":{"title":"Seg-R1: Segmentation Can Be Surprisingly Simple with Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Zuxuan Wu, Zuyao You","submitted_at":"2025-06-27T20:40:45Z","abstract_excerpt":"We present Seg-R1, a preliminary exploration of using reinforcement learning (RL) to enhance the pixel-level understanding and reasoning capabilities of large multimodal models (LMMs). Starting with foreground segmentation tasks, specifically camouflaged object detection (COD) and salient object detection (SOD), our approach enables the LMM to generate point and bounding box prompts in the next-token fashion, which are then used to guide SAM2 in producing segmentation masks. We introduce Group Relative Policy Optimization (GRPO) into the segmentation domain, equipping the LMM with pixel-level "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.22624","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-27T20:40:45Z","cross_cats_sorted":[],"title_canon_sha256":"4ca69c3921d2a7d2a059de4d25f4117dbe28a1a73b5305f01a25c72cf56c43e5","abstract_canon_sha256":"79abc1f80b90e73737d02ada2b89e0f19d37358c97c274fd05915d0b8a734aad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:42.114345Z","signature_b64":"E1OMv/gXswktTdRA5MJY1/6wYQzEm9ABVLnOZLbwCgTEL11uvvH92p5lTTDa+zFolfTJulmDatL8nse1MEv1BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"624a945445a324fbb270abd02b5c4bedbdc7196e9cb9d563b99505c684deb77f","last_reissued_at":"2026-07-05T11:28:42.113825Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:42.113825Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Seg-R1: Segmentation Can Be Surprisingly Simple with Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Zuxuan Wu, Zuyao You","submitted_at":"2025-06-27T20:40:45Z","abstract_excerpt":"We present Seg-R1, a preliminary exploration of using reinforcement learning (RL) to enhance the pixel-level understanding and reasoning capabilities of large multimodal models (LMMs). Starting with foreground segmentation tasks, specifically camouflaged object detection (COD) and salient object detection (SOD), our approach enables the LMM to generate point and bounding box prompts in the next-token fashion, which are then used to guide SAM2 in producing segmentation masks. We introduce Group Relative Policy Optimization (GRPO) into the segmentation domain, equipping the LMM with pixel-level "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.22624","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.22624/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.22624","created_at":"2026-07-05T11:28:42.113886+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.22624v1","created_at":"2026-07-05T11:28:42.113886+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.22624","created_at":"2026-07-05T11:28:42.113886+00:00"},{"alias_kind":"pith_short_12","alias_value":"MJFJIVCFUMSP","created_at":"2026-07-05T11:28:42.113886+00:00"},{"alias_kind":"pith_short_16","alias_value":"MJFJIVCFUMSPXMTQ","created_at":"2026-07-05T11:28:42.113886+00:00"},{"alias_kind":"pith_short_8","alias_value":"MJFJIVCF","created_at":"2026-07-05T11:28:42.113886+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09303","citing_title":"Reason Twice: Segmentation via Candidate Discovery and Comparative Reasoning","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31924","citing_title":"InstanceControl: Controllable Complex Image Generation without Instance Labeling","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14742","citing_title":"EARL: Towards a Unified Analysis-Guided Reinforcement Learning Framework for Egocentric Interaction Reasoning and Pixel Grounding","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20385","citing_title":"ConceptSeg-R1: Segment Any Concept via Meta-Reinforcement Learning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15951","citing_title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03043","citing_title":"OneThinker: All-in-one Reasoning Model for Image and Video","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10554","citing_title":"Grounding Everything in Tokens for Multimodal Large Language Models","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00181","citing_title":"CamReasoner: Reinforcing Camera Movement Understanding via Structured Spatial Reasoning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22844","citing_title":"PhySe-RPO: Physics and Semantics Guided Relative Policy Optimization for Diffusion-Based Surgical Smoke Removal","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12497","citing_title":"From Web to Pixels: Bringing Agentic Search into Visual Perception","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W","json":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W.json","graph_json":"https://pith.science/api/pith-number/MJFJIVCFUMSPXMTQVPICWXCL5W/graph.json","events_json":"https://pith.science/api/pith-number/MJFJIVCFUMSPXMTQVPICWXCL5W/events.json","paper":"https://pith.science/paper/MJFJIVCF"},"agent_actions":{"view_html":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W","download_json":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W.json","view_paper":"https://pith.science/paper/MJFJIVCF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.22624&json=true","fetch_graph":"https://pith.science/api/pith-number/MJFJIVCFUMSPXMTQVPICWXCL5W/graph.json","fetch_events":"https://pith.science/api/pith-number/MJFJIVCFUMSPXMTQVPICWXCL5W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W/action/storage_attestation","attest_author":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W/action/author_attestation","sign_citation":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W/action/citation_signature","submit_replication":"https://pith.science/pith/MJFJIVCFUMSPXMTQVPICWXCL5W/action/replication_record"}},"created_at":"2026-07-05T11:28:42.113886+00:00","updated_at":"2026-07-05T11:28:42.113886+00:00"}