{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2IE7GBSSJXC3HB3OAX6OQEPARH","short_pith_number":"pith:2IE7GBSS","schema_version":"1.0","canonical_sha256":"d209f306524dc5b3876e05fce811e089c2c7cca932c5bbbcf8c609e35d76f875","source":{"kind":"arxiv","id":"2310.03720","version":4},"attestation_state":"computed","paper":{"title":"SteP: Stacked LLM Policies for Web Actions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Paloma Sodhi, Ryan McDonald, S.R.K. Branavan, Yoav Artzi","submitted_at":"2023-10-05T17:40:09Z","abstract_excerpt":"Performing tasks on the web presents fundamental challenges to large language models (LLMs), including combinatorially large open-world tasks and variations across web interfaces. Simply specifying a large prompt to handle all possible behaviors and states is extremely complex, and results in behavior leaks between unrelated behaviors. Decomposition to distinct policies can address this challenge, but requires carefully handing off control between policies. We propose Stacked LLM Policies for Web Actions (SteP), an approach to dynamically compose policies to solve a diverse set of web tasks. S"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03720","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-05T17:40:09Z","cross_cats_sorted":[],"title_canon_sha256":"f48ebcece83a3e646c912e78dec3dbc41394083953d1ea7caab20f677d902575","abstract_canon_sha256":"593d64428af15a71137679b7c34d50fec82d645978e25f667f51356d0cbf8fbb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:53:38.070494Z","signature_b64":"bzsw3dQxxvd0GG/F6T3Aq2IHJGYu+haErdGYfsWP3ENrfF6OazkXbk2AchAC9XjeVoQVa5jbOpJZCanWT0y5CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d209f306524dc5b3876e05fce811e089c2c7cca932c5bbbcf8c609e35d76f875","last_reissued_at":"2026-07-05T08:53:38.070080Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:53:38.070080Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SteP: Stacked LLM Policies for Web Actions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Paloma Sodhi, Ryan McDonald, S.R.K. Branavan, Yoav Artzi","submitted_at":"2023-10-05T17:40:09Z","abstract_excerpt":"Performing tasks on the web presents fundamental challenges to large language models (LLMs), including combinatorially large open-world tasks and variations across web interfaces. Simply specifying a large prompt to handle all possible behaviors and states is extremely complex, and results in behavior leaks between unrelated behaviors. Decomposition to distinct policies can address this challenge, but requires carefully handing off control between policies. We propose Stacked LLM Policies for Web Actions (SteP), an approach to dynamically compose policies to solve a diverse set of web tasks. S"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03720","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03720/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03720","created_at":"2026-07-05T08:53:38.070134+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03720v4","created_at":"2026-07-05T08:53:38.070134+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03720","created_at":"2026-07-05T08:53:38.070134+00:00"},{"alias_kind":"pith_short_12","alias_value":"2IE7GBSSJXC3","created_at":"2026-07-05T08:53:38.070134+00:00"},{"alias_kind":"pith_short_16","alias_value":"2IE7GBSSJXC3HB3O","created_at":"2026-07-05T08:53:38.070134+00:00"},{"alias_kind":"pith_short_8","alias_value":"2IE7GBSS","created_at":"2026-07-05T08:53:38.070134+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20683","citing_title":"From Question Answering to Task Completion: A Survey on Agent System and Harness Design","ref_index":252,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23939","citing_title":"DRIVE: Modeling Skills at the Reasoning and Interaction Levels for Web Agents under Continual Learning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09572","citing_title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2409.07429","citing_title":"Agent Workflow Memory","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH","json":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH.json","graph_json":"https://pith.science/api/pith-number/2IE7GBSSJXC3HB3OAX6OQEPARH/graph.json","events_json":"https://pith.science/api/pith-number/2IE7GBSSJXC3HB3OAX6OQEPARH/events.json","paper":"https://pith.science/paper/2IE7GBSS"},"agent_actions":{"view_html":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH","download_json":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH.json","view_paper":"https://pith.science/paper/2IE7GBSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03720&json=true","fetch_graph":"https://pith.science/api/pith-number/2IE7GBSSJXC3HB3OAX6OQEPARH/graph.json","fetch_events":"https://pith.science/api/pith-number/2IE7GBSSJXC3HB3OAX6OQEPARH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH/action/storage_attestation","attest_author":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH/action/author_attestation","sign_citation":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH/action/citation_signature","submit_replication":"https://pith.science/pith/2IE7GBSSJXC3HB3OAX6OQEPARH/action/replication_record"}},"created_at":"2026-07-05T08:53:38.070134+00:00","updated_at":"2026-07-05T08:53:38.070134+00:00"}