{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:GS467ITWDDXN36K4C6BXWWL4TW","short_pith_number":"pith:GS467ITW","schema_version":"1.0","canonical_sha256":"34b9efa27618eeddf95c17837b597c9daa85d39948d9447e13c7626ed1dab163","source":{"kind":"arxiv","id":"2203.07540","version":2},"attestation_state":"computed","paper":{"title":"ScienceWorld: Is your Agent Smarter than a 5th Grader?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Marc-Alexandre C\\^ot\\'e, Peter Jansen, Prithviraj Ammanabrolu, Ruoyao Wang","submitted_at":"2022-03-14T22:52:34Z","abstract_excerpt":"We present ScienceWorld, a benchmark to test agents' scientific reasoning abilities in a new interactive text environment at the level of a standard elementary school science curriculum. Despite the transformer-based progress seen in question-answering and scientific text processing, we find that current models cannot reason about or explain learned science concepts in novel contexts. For instance, models can easily answer what the conductivity of a known material is but struggle when asked how they would conduct an experiment in a grounded environment to find the conductivity of an unknown ma"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.07540","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-03-14T22:52:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e23af4b8e860e7886ee50d0757a09b68cdbdc7cbc6f20f79cce6163bf560af72","abstract_canon_sha256":"b7baf3b8a31069f667583a36b366e44104ada49c87bf128d875df04018f4edc8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:15:37.763461Z","signature_b64":"01AfjVRfharo4uU6SyfGiLKeCsZmJYzN1aAomVNbjGTOixKEDBxlRZ3xLqpNxroQ3ygG60I3ZLn9wrmbHQdgAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34b9efa27618eeddf95c17837b597c9daa85d39948d9447e13c7626ed1dab163","last_reissued_at":"2026-07-05T05:15:37.762948Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:15:37.762948Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ScienceWorld: Is your Agent Smarter than a 5th Grader?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Marc-Alexandre C\\^ot\\'e, Peter Jansen, Prithviraj Ammanabrolu, Ruoyao Wang","submitted_at":"2022-03-14T22:52:34Z","abstract_excerpt":"We present ScienceWorld, a benchmark to test agents' scientific reasoning abilities in a new interactive text environment at the level of a standard elementary school science curriculum. Despite the transformer-based progress seen in question-answering and scientific text processing, we find that current models cannot reason about or explain learned science concepts in novel contexts. For instance, models can easily answer what the conductivity of a known material is but struggle when asked how they would conduct an experiment in a grounded environment to find the conductivity of an unknown ma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.07540","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.07540/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.07540","created_at":"2026-07-05T05:15:37.763015+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.07540v2","created_at":"2026-07-05T05:15:37.763015+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.07540","created_at":"2026-07-05T05:15:37.763015+00:00"},{"alias_kind":"pith_short_12","alias_value":"GS467ITWDDXN","created_at":"2026-07-05T05:15:37.763015+00:00"},{"alias_kind":"pith_short_16","alias_value":"GS467ITWDDXN36K4","created_at":"2026-07-05T05:15:37.763015+00:00"},{"alias_kind":"pith_short_8","alias_value":"GS467ITW","created_at":"2026-07-05T05:15:37.763015+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05339","citing_title":"TREK: Distill to Explore, Reinforce to Refine","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26935","citing_title":"Where Do CoT Training Gains Land in LLM based Agents?","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26918","citing_title":"Diagnosing Task Insensitivity in Language Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19409","citing_title":"OpenRath: Session-Centered Runtime State for Agent Systems","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29537","citing_title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03715","citing_title":"R$^3$L: Reflect-then-Retry Reinforcement Learning with Language-Guided Exploration, Pivotal Credit, and Positive Amplification","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16198","citing_title":"Formal Methods Meet LLMs: Auditing, Monitoring, and Intervention for Compliance of Advanced AI Systems","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13399","citing_title":"Differentiable Evolutionary Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09514","citing_title":"EcoGym: Evaluating LLMs for Long-Horizon Plan-and-Execute in Interactive Economies","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2402.02716","citing_title":"Understanding the planning of LLM agents: A survey","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23194","citing_title":"From Coarse to Fine: Self-Adaptive Hierarchical Planning for LLM Agents","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW","json":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW.json","graph_json":"https://pith.science/api/pith-number/GS467ITWDDXN36K4C6BXWWL4TW/graph.json","events_json":"https://pith.science/api/pith-number/GS467ITWDDXN36K4C6BXWWL4TW/events.json","paper":"https://pith.science/paper/GS467ITW"},"agent_actions":{"view_html":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW","download_json":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW.json","view_paper":"https://pith.science/paper/GS467ITW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.07540&json=true","fetch_graph":"https://pith.science/api/pith-number/GS467ITWDDXN36K4C6BXWWL4TW/graph.json","fetch_events":"https://pith.science/api/pith-number/GS467ITWDDXN36K4C6BXWWL4TW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW/action/storage_attestation","attest_author":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW/action/author_attestation","sign_citation":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW/action/citation_signature","submit_replication":"https://pith.science/pith/GS467ITWDDXN36K4C6BXWWL4TW/action/replication_record"}},"created_at":"2026-07-05T05:15:37.763015+00:00","updated_at":"2026-07-05T05:15:37.763015+00:00"}