{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:G3MPSI4ISQQKVE6NR3YRYTKHHZ","short_pith_number":"pith:G3MPSI4I","schema_version":"1.0","canonical_sha256":"36d8f923889420aa93cd8ef11c4d473e6869020b375631f7a23e5aba080453f2","source":{"kind":"arxiv","id":"2505.03947","version":1},"attestation_state":"computed","paper":{"title":"Frog Soup: Zero-Shot, In-Context, and Sample-Efficient Frogger Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Doug Fulop, Xiang Li, Yiyang Hao","submitted_at":"2025-05-06T19:51:41Z","abstract_excerpt":"One of the primary aspirations in reinforcement learning research is developing general-purpose agents capable of rapidly adapting to and mastering novel tasks. While RL gaming agents have mastered many Atari games, they remain slow and costly to train for each game. In this work, we demonstrate that latest reasoning LLMs with out-of-domain RL post-training can play a challenging Atari game called Frogger under a zero-shot setting. We then investigate the effect of in-context learning and the amount of reasoning effort on LLM performance. Lastly, we demonstrate a way to bootstrap traditional R"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.03947","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-06T19:51:41Z","cross_cats_sorted":[],"title_canon_sha256":"05069ef467799daee9d324a742d1e0832469bfc6bf6a0bdc0f213ca7dba423fb","abstract_canon_sha256":"586afc73865fd50e290999c52f2cd72cab82756dc30f7e6236d8aa0cd6007978"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:34.037800Z","signature_b64":"HHp6aFLA5RZaw+KGEoZAS3ER4QnhUs1FfOYkW9t7IhD+wpICR21bAvnural8qNOBn0xZvqFciRF+n+gKYK9NBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36d8f923889420aa93cd8ef11c4d473e6869020b375631f7a23e5aba080453f2","last_reissued_at":"2026-07-05T10:59:34.037370Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:34.037370Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Frog Soup: Zero-Shot, In-Context, and Sample-Efficient Frogger Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Doug Fulop, Xiang Li, Yiyang Hao","submitted_at":"2025-05-06T19:51:41Z","abstract_excerpt":"One of the primary aspirations in reinforcement learning research is developing general-purpose agents capable of rapidly adapting to and mastering novel tasks. While RL gaming agents have mastered many Atari games, they remain slow and costly to train for each game. In this work, we demonstrate that latest reasoning LLMs with out-of-domain RL post-training can play a challenging Atari game called Frogger under a zero-shot setting. We then investigate the effect of in-context learning and the amount of reasoning effort on LLM performance. Lastly, we demonstrate a way to bootstrap traditional R"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.03947","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.03947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.03947","created_at":"2026-07-05T10:59:34.037431+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.03947v1","created_at":"2026-07-05T10:59:34.037431+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.03947","created_at":"2026-07-05T10:59:34.037431+00:00"},{"alias_kind":"pith_short_12","alias_value":"G3MPSI4ISQQK","created_at":"2026-07-05T10:59:34.037431+00:00"},{"alias_kind":"pith_short_16","alias_value":"G3MPSI4ISQQKVE6N","created_at":"2026-07-05T10:59:34.037431+00:00"},{"alias_kind":"pith_short_8","alias_value":"G3MPSI4I","created_at":"2026-07-05T10:59:34.037431+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ","json":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ.json","graph_json":"https://pith.science/api/pith-number/G3MPSI4ISQQKVE6NR3YRYTKHHZ/graph.json","events_json":"https://pith.science/api/pith-number/G3MPSI4ISQQKVE6NR3YRYTKHHZ/events.json","paper":"https://pith.science/paper/G3MPSI4I"},"agent_actions":{"view_html":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ","download_json":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ.json","view_paper":"https://pith.science/paper/G3MPSI4I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.03947&json=true","fetch_graph":"https://pith.science/api/pith-number/G3MPSI4ISQQKVE6NR3YRYTKHHZ/graph.json","fetch_events":"https://pith.science/api/pith-number/G3MPSI4ISQQKVE6NR3YRYTKHHZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ/action/storage_attestation","attest_author":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ/action/author_attestation","sign_citation":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ/action/citation_signature","submit_replication":"https://pith.science/pith/G3MPSI4ISQQKVE6NR3YRYTKHHZ/action/replication_record"}},"created_at":"2026-07-05T10:59:34.037431+00:00","updated_at":"2026-07-05T10:59:34.037431+00:00"}