{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DW2QGAHFFF6TGQK7HXJQITOXVE","short_pith_number":"pith:DW2QGAHF","schema_version":"1.0","canonical_sha256":"1db50300e5297d33415f3dd3044dd7a921dfaca99c52c09fd219a6b9b80dc90c","source":{"kind":"arxiv","id":"2505.18121","version":1},"attestation_state":"computed","paper":{"title":"ProgRM: Build Better GUI Agents with Progress Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Danyang Zhang, Kai Yu, Lu Chen, Ruisheng Cao, Situo Zhang, Zichen Zhu, Zihan Zhao, Ziyue Yang","submitted_at":"2025-05-23T17:23:11Z","abstract_excerpt":"LLM-based (Large Language Model) GUI (Graphical User Interface) agents can potentially reshape our daily lives significantly. However, current LLM-based GUI agents suffer from the scarcity of high-quality training data owing to the difficulties of trajectory collection and reward annotation. Existing works have been exploring LLMs to collect trajectories for imitation learning or to offer reward signals for online RL training. However, the Outcome Reward Model (ORM) used in existing works cannot provide finegrained feedback and can over-penalize the valuable steps in finally failed trajectorie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18121","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-23T17:23:11Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"9cc6d25bdcde2908290eab28cb07de2eaae5cb94a955f4f4144d8f54db0c353f","abstract_canon_sha256":"7fc1ac5256ebac77da1c51925cd5c660fc08c9f5518b4a81d55e0af3867c6005"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:36.928208Z","signature_b64":"3dzooAY6yz9PxlyH1yPEay5kW7T3E5hKTKa/waE2lYn1W+AGg2nyxZQkR6nX3aTUGSQKxv3Y1E053Ik99svGBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1db50300e5297d33415f3dd3044dd7a921dfaca99c52c09fd219a6b9b80dc90c","last_reissued_at":"2026-07-05T11:08:36.927740Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:36.927740Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ProgRM: Build Better GUI Agents with Progress Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Danyang Zhang, Kai Yu, Lu Chen, Ruisheng Cao, Situo Zhang, Zichen Zhu, Zihan Zhao, Ziyue Yang","submitted_at":"2025-05-23T17:23:11Z","abstract_excerpt":"LLM-based (Large Language Model) GUI (Graphical User Interface) agents can potentially reshape our daily lives significantly. However, current LLM-based GUI agents suffer from the scarcity of high-quality training data owing to the difficulties of trajectory collection and reward annotation. Existing works have been exploring LLMs to collect trajectories for imitation learning or to offer reward signals for online RL training. However, the Outcome Reward Model (ORM) used in existing works cannot provide finegrained feedback and can over-penalize the valuable steps in finally failed trajectorie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18121","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18121/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18121","created_at":"2026-07-05T11:08:36.927798+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18121v1","created_at":"2026-07-05T11:08:36.927798+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18121","created_at":"2026-07-05T11:08:36.927798+00:00"},{"alias_kind":"pith_short_12","alias_value":"DW2QGAHFFF6T","created_at":"2026-07-05T11:08:36.927798+00:00"},{"alias_kind":"pith_short_16","alias_value":"DW2QGAHFFF6TGQK7","created_at":"2026-07-05T11:08:36.927798+00:00"},{"alias_kind":"pith_short_8","alias_value":"DW2QGAHF","created_at":"2026-07-05T11:08:36.927798+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07027","citing_title":"StainFlow: Entity-Stain Tracking and Evidence Linking for Process Rewards in GUI Agents","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00564","citing_title":"Decomposed On-Policy Distillation for Vision-Language Reasoning: Steering Gradients for Visual Grounding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02345","citing_title":"UI-Oceanus: Scaling GUI Agents with Synthetic Environmental Dynamics","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27955","citing_title":"GUI Agents with Reinforcement Learning: Toward Digital Inhabitants","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07110","citing_title":"Securing Computer-Use Agents: A Unified Architecture-Lifecycle Framework for Deployment-Grounded Reliability","ref_index":114,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE","json":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE.json","graph_json":"https://pith.science/api/pith-number/DW2QGAHFFF6TGQK7HXJQITOXVE/graph.json","events_json":"https://pith.science/api/pith-number/DW2QGAHFFF6TGQK7HXJQITOXVE/events.json","paper":"https://pith.science/paper/DW2QGAHF"},"agent_actions":{"view_html":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE","download_json":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE.json","view_paper":"https://pith.science/paper/DW2QGAHF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18121&json=true","fetch_graph":"https://pith.science/api/pith-number/DW2QGAHFFF6TGQK7HXJQITOXVE/graph.json","fetch_events":"https://pith.science/api/pith-number/DW2QGAHFFF6TGQK7HXJQITOXVE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE/action/storage_attestation","attest_author":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE/action/author_attestation","sign_citation":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE/action/citation_signature","submit_replication":"https://pith.science/pith/DW2QGAHFFF6TGQK7HXJQITOXVE/action/replication_record"}},"created_at":"2026-07-05T11:08:36.927798+00:00","updated_at":"2026-07-05T11:08:36.927798+00:00"}