{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RKA3UAHLH6XX2KYEXS3LU4WQ7N","short_pith_number":"pith:RKA3UAHL","schema_version":"1.0","canonical_sha256":"8a81ba00eb3faf7d2b04bcb6ba72d0fb7c2d6cb9c5601c5cdbd697960aa97cda","source":{"kind":"arxiv","id":"2405.11120","version":1},"attestation_state":"computed","paper":{"title":"Latent State Estimation Helps UI Agents to Reason","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Alice Li, Christopher Rawles, Oriana Riva, William E Bishop","submitted_at":"2024-05-17T23:27:33Z","abstract_excerpt":"A common problem for agents operating in real-world environments is that the response of an environment to their actions may be non-deterministic and observed through noise. This renders environmental state and progress towards completing a task latent. Despite recent impressive demonstrations of LLM's reasoning abilities on various benchmarks, whether LLMs can build estimates of latent state and leverage them for reasoning has not been explicitly studied. We investigate this problem in the real-world domain of autonomous UI agents. We establish that appropriately prompting LLMs in a zero-shot"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.11120","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-17T23:27:33Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e7317aef69b69106742ef48f6a6e25dda1d7713dd5f5cb67bdd411d8923f4a14","abstract_canon_sha256":"99de2f2c68961271aac2ce27dac9235f9ac7f338dd8fd40ae3ffac585d8a7e26"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:33.553873Z","signature_b64":"jACP1kTfjR9bX6SWsRMmns+jBiPN63Z8kM4q1IMy70F9Ugf3oakOTmKB6+WgNHdxmTARjqQrBbUaAOTSWozNBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a81ba00eb3faf7d2b04bcb6ba72d0fb7c2d6cb9c5601c5cdbd697960aa97cda","last_reissued_at":"2026-07-05T08:20:33.553400Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:33.553400Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Latent State Estimation Helps UI Agents to Reason","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Alice Li, Christopher Rawles, Oriana Riva, William E Bishop","submitted_at":"2024-05-17T23:27:33Z","abstract_excerpt":"A common problem for agents operating in real-world environments is that the response of an environment to their actions may be non-deterministic and observed through noise. This renders environmental state and progress towards completing a task latent. Despite recent impressive demonstrations of LLM's reasoning abilities on various benchmarks, whether LLMs can build estimates of latent state and leverage them for reasoning has not been explicitly studied. We investigate this problem in the real-world domain of autonomous UI agents. We establish that appropriately prompting LLMs in a zero-shot"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.11120","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.11120/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.11120","created_at":"2026-07-05T08:20:33.553459+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.11120v1","created_at":"2026-07-05T08:20:33.553459+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.11120","created_at":"2026-07-05T08:20:33.553459+00:00"},{"alias_kind":"pith_short_12","alias_value":"RKA3UAHLH6XX","created_at":"2026-07-05T08:20:33.553459+00:00"},{"alias_kind":"pith_short_16","alias_value":"RKA3UAHLH6XX2KYE","created_at":"2026-07-05T08:20:33.553459+00:00"},{"alias_kind":"pith_short_8","alias_value":"RKA3UAHL","created_at":"2026-07-05T08:20:33.553459+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N","json":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N.json","graph_json":"https://pith.science/api/pith-number/RKA3UAHLH6XX2KYEXS3LU4WQ7N/graph.json","events_json":"https://pith.science/api/pith-number/RKA3UAHLH6XX2KYEXS3LU4WQ7N/events.json","paper":"https://pith.science/paper/RKA3UAHL"},"agent_actions":{"view_html":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N","download_json":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N.json","view_paper":"https://pith.science/paper/RKA3UAHL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.11120&json=true","fetch_graph":"https://pith.science/api/pith-number/RKA3UAHLH6XX2KYEXS3LU4WQ7N/graph.json","fetch_events":"https://pith.science/api/pith-number/RKA3UAHLH6XX2KYEXS3LU4WQ7N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N/action/storage_attestation","attest_author":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N/action/author_attestation","sign_citation":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N/action/citation_signature","submit_replication":"https://pith.science/pith/RKA3UAHLH6XX2KYEXS3LU4WQ7N/action/replication_record"}},"created_at":"2026-07-05T08:20:33.553459+00:00","updated_at":"2026-07-05T08:20:33.553459+00:00"}