{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UGQICNMEJVLJRLBJZCHMUOGAQM","short_pith_number":"pith:UGQICNME","schema_version":"1.0","canonical_sha256":"a1a08135844d5698ac29c88eca38c0833b2136e119660732a02db766833186a2","source":{"kind":"arxiv","id":"2404.06474","version":3},"attestation_state":"computed","paper":{"title":"Autonomous Evaluation and Refinement of Digital Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alane Suhr, Jiayi Pan, Nicholas Tomlin, Sergey Levine, Yichi Zhang, Yifei Zhou","submitted_at":"2024-04-09T17:25:47Z","abstract_excerpt":"We show that domain-general automatic evaluators can significantly improve the performance of agents for web navigation and device control. We experiment with multiple evaluation models that trade off between inference cost, modularity of design, and accuracy. We validate the performance of these models in several popular benchmarks for digital agents, finding between 74.4 and 92.9% agreement with oracle evaluation metrics. Finally, we use these evaluators to improve the performance of existing agents via fine-tuning and inference-time guidance. Without any additional supervision, we improve s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.06474","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-04-09T17:25:47Z","cross_cats_sorted":[],"title_canon_sha256":"99db48b3cd32ce410a1846abd0db88a61fa2a5849411f03245a0b7134b642c2d","abstract_canon_sha256":"700f86aec10e4c7f50823b3042c6cb13d7ab5746a48385e18109a6069a963234"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:48.165331Z","signature_b64":"eMCIk05ZDmndRaAuvFarVoQ9Ft4WGcUFu2XouP69Y889lM+iBescCXfqSON6yHypUf0FpNKz/Z5cn+jSvW5zCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a1a08135844d5698ac29c88eca38c0833b2136e119660732a02db766833186a2","last_reissued_at":"2026-07-05T09:16:48.164800Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:48.164800Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Autonomous Evaluation and Refinement of Digital Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alane Suhr, Jiayi Pan, Nicholas Tomlin, Sergey Levine, Yichi Zhang, Yifei Zhou","submitted_at":"2024-04-09T17:25:47Z","abstract_excerpt":"We show that domain-general automatic evaluators can significantly improve the performance of agents for web navigation and device control. We experiment with multiple evaluation models that trade off between inference cost, modularity of design, and accuracy. We validate the performance of these models in several popular benchmarks for digital agents, finding between 74.4 and 92.9% agreement with oracle evaluation metrics. Finally, we use these evaluators to improve the performance of existing agents via fine-tuning and inference-time guidance. Without any additional supervision, we improve s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.06474","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.06474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.06474","created_at":"2026-07-05T09:16:48.164862+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.06474v3","created_at":"2026-07-05T09:16:48.164862+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.06474","created_at":"2026-07-05T09:16:48.164862+00:00"},{"alias_kind":"pith_short_12","alias_value":"UGQICNMEJVLJ","created_at":"2026-07-05T09:16:48.164862+00:00"},{"alias_kind":"pith_short_16","alias_value":"UGQICNMEJVLJRLBJ","created_at":"2026-07-05T09:16:48.164862+00:00"},{"alias_kind":"pith_short_8","alias_value":"UGQICNME","created_at":"2026-07-05T09:16:48.164862+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06118","citing_title":"WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21654","citing_title":"ChainWorld: Composing Long-Horizon Desktop Workloads from Atomic OSWorld Tasks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17628","citing_title":"OPD-Evolver: Cultivating Holistic Agent Evolver via On-Policy Distillation","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06462","citing_title":"Benchmark Everything Everywhere All at Once","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22564","citing_title":"SynAE: A Framework for Measuring the Quality of Synthetic Data for Tool-Calling Agent Evaluations","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18660","citing_title":"Evaluating Multi-turn Human-AI Interaction","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09572","citing_title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19396","citing_title":"EchoTrail-GUI: Building Actionable Memory for GUI Agents via Critic-Guided Self-Exploration","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15808","citing_title":"Inference-Time Scaling of Verification: Self-Evolving Deep Research Agents via Test-Time Rubric-Guided Verification","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2409.07429","citing_title":"Agent Workflow Memory","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04399","citing_title":"GUIDE: Interpretable GUI Agent Evaluation via Hierarchical Diagnosis","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM","json":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM.json","graph_json":"https://pith.science/api/pith-number/UGQICNMEJVLJRLBJZCHMUOGAQM/graph.json","events_json":"https://pith.science/api/pith-number/UGQICNMEJVLJRLBJZCHMUOGAQM/events.json","paper":"https://pith.science/paper/UGQICNME"},"agent_actions":{"view_html":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM","download_json":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM.json","view_paper":"https://pith.science/paper/UGQICNME","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.06474&json=true","fetch_graph":"https://pith.science/api/pith-number/UGQICNMEJVLJRLBJZCHMUOGAQM/graph.json","fetch_events":"https://pith.science/api/pith-number/UGQICNMEJVLJRLBJZCHMUOGAQM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM/action/storage_attestation","attest_author":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM/action/author_attestation","sign_citation":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM/action/citation_signature","submit_replication":"https://pith.science/pith/UGQICNMEJVLJRLBJZCHMUOGAQM/action/replication_record"}},"created_at":"2026-07-05T09:16:48.164862+00:00","updated_at":"2026-07-05T09:16:48.164862+00:00"}