{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YSBWPNCLMMWDTZYARYQOTVOFU3","short_pith_number":"pith:YSBWPNCL","schema_version":"1.0","canonical_sha256":"c48367b44b632c39e7008e20e9d5c5a6d31d96339105526fbf40a8dc600b48f3","source":{"kind":"arxiv","id":"2508.04389","version":1},"attestation_state":"computed","paper":{"title":"GuirlVG: Incentivize GUI Visual Grounding via Empirical Exploration on Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bin Lei, Caiwen Ding, Gaowen Liu, Weitai Kang, Yan Yan","submitted_at":"2025-08-06T12:35:24Z","abstract_excerpt":"Graphical user interface visual grounding (GUI-VG), a core capability for GUI agents, has primarily relied on supervised fine-tuning (SFT) of multimodal large language models (MLLMs), which demands extensive data curation and significant training costs. However, as MLLMs continue to advance and even cover GUI domains during pretraining, the necessity of exhaustive SFT post-training becomes increasingly questionable. Meanwhile, recent successes of rule-based reinforcement fine-tuning (RFT) suggest a more efficient alternative. Despite this promise, the optimal manner of applying RFT for GUI-VG "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.04389","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-08-06T12:35:24Z","cross_cats_sorted":[],"title_canon_sha256":"e836679ec34a4da0b3bc114baf1cc8256d186b63c04383fe9ec04d84553c7b47","abstract_canon_sha256":"9464efc40a1e53a47d2d21bf32f22c3c7ee0c9035d8c10eafc46cba98d8f51fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:25.527490Z","signature_b64":"kMsdY8pFrGPRzc6H/zCSg9wwmwEpsPSVB9HLPgHP8uCp/UGZpHdhAZ6wSZGKITflCKDXWaKeA69wKor5gMpYDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c48367b44b632c39e7008e20e9d5c5a6d31d96339105526fbf40a8dc600b48f3","last_reissued_at":"2026-07-05T11:49:25.526983Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:25.526983Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GuirlVG: Incentivize GUI Visual Grounding via Empirical Exploration on Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bin Lei, Caiwen Ding, Gaowen Liu, Weitai Kang, Yan Yan","submitted_at":"2025-08-06T12:35:24Z","abstract_excerpt":"Graphical user interface visual grounding (GUI-VG), a core capability for GUI agents, has primarily relied on supervised fine-tuning (SFT) of multimodal large language models (MLLMs), which demands extensive data curation and significant training costs. However, as MLLMs continue to advance and even cover GUI domains during pretraining, the necessity of exhaustive SFT post-training becomes increasingly questionable. Meanwhile, recent successes of rule-based reinforcement fine-tuning (RFT) suggest a more efficient alternative. Despite this promise, the optimal manner of applying RFT for GUI-VG "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.04389","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.04389/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.04389","created_at":"2026-07-05T11:49:25.527046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.04389v1","created_at":"2026-07-05T11:49:25.527046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.04389","created_at":"2026-07-05T11:49:25.527046+00:00"},{"alias_kind":"pith_short_12","alias_value":"YSBWPNCLMMWD","created_at":"2026-07-05T11:49:25.527046+00:00"},{"alias_kind":"pith_short_16","alias_value":"YSBWPNCLMMWDTZYA","created_at":"2026-07-05T11:49:25.527046+00:00"},{"alias_kind":"pith_short_8","alias_value":"YSBWPNCL","created_at":"2026-07-05T11:49:25.527046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30084","citing_title":"One Forward Beats Two: InnerZoom for Accurate and Efficient GUI Grounding","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00642","citing_title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00642","citing_title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21268","citing_title":"Measure Twice, Click Once: Co-evolving Proposer and Visual Critic via Reinforcement Learning for GUI Grounding","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14262","citing_title":"GUI-Perturbed: Domain Randomization Reveals Systematic Brittleness in GUI Grounding Models","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3","json":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3.json","graph_json":"https://pith.science/api/pith-number/YSBWPNCLMMWDTZYARYQOTVOFU3/graph.json","events_json":"https://pith.science/api/pith-number/YSBWPNCLMMWDTZYARYQOTVOFU3/events.json","paper":"https://pith.science/paper/YSBWPNCL"},"agent_actions":{"view_html":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3","download_json":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3.json","view_paper":"https://pith.science/paper/YSBWPNCL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.04389&json=true","fetch_graph":"https://pith.science/api/pith-number/YSBWPNCLMMWDTZYARYQOTVOFU3/graph.json","fetch_events":"https://pith.science/api/pith-number/YSBWPNCLMMWDTZYARYQOTVOFU3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3/action/storage_attestation","attest_author":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3/action/author_attestation","sign_citation":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3/action/citation_signature","submit_replication":"https://pith.science/pith/YSBWPNCLMMWDTZYARYQOTVOFU3/action/replication_record"}},"created_at":"2026-07-05T11:49:25.527046+00:00","updated_at":"2026-07-05T11:49:25.527046+00:00"}