{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:TWQVOG44RPEQVWTVRC5TGG6IIO","short_pith_number":"pith:TWQVOG44","schema_version":"1.0","canonical_sha256":"9da1571b9c8bc90ada7588bb331bc84395aeaca80f464bb4a63b8f031cd5f67a","source":{"kind":"arxiv","id":"2607.17760","version":1},"attestation_state":"computed","paper":{"title":"Generalize and Guide: Decomposing Rewards for Few-Shot Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Grace Zhang, Ziyi Liu","submitted_at":"2026-07-20T09:51:02Z","abstract_excerpt":"Inverse reinforcement learning (IRL) provides a powerful framework for learning from demonstrations. However, real-world tasks often exhibit substantial natural variations (e.g., picking up mugs with varying shapes), making it impractical to collect demonstrations that fully specify a new task under every possible scenario. In practice, while demonstrations for the target task are limited, it is often easier to obtain datasets of heterogeneous but related behaviors. This motivates the problem of few-shot IRL with multi-task demonstrations (FM-IRL), where an agent must learn a new task with sub"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.17760","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-20T09:51:02Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"00a73b15a52438473a87125b6bbda311e0ae0a87868f39de5d2638a621a9bfe6","abstract_canon_sha256":"152ed55722c91a292e227bbbba3903fe866ba41ffad8fa8177493c8982f3ea52"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T02:21:58.780239Z","signature_b64":"kT/gR6ZTPkfxtuQls7f9Fh+vxS+Rd7Ngv42eWIwu7jmjHFvaW3xEtNtjn2qFNM0pf4KAOmv6ZCACEGTwziJhBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9da1571b9c8bc90ada7588bb331bc84395aeaca80f464bb4a63b8f031cd5f67a","last_reissued_at":"2026-07-21T02:21:58.779437Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T02:21:58.779437Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalize and Guide: Decomposing Rewards for Few-Shot Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Grace Zhang, Ziyi Liu","submitted_at":"2026-07-20T09:51:02Z","abstract_excerpt":"Inverse reinforcement learning (IRL) provides a powerful framework for learning from demonstrations. However, real-world tasks often exhibit substantial natural variations (e.g., picking up mugs with varying shapes), making it impractical to collect demonstrations that fully specify a new task under every possible scenario. In practice, while demonstrations for the target task are limited, it is often easier to obtain datasets of heterogeneous but related behaviors. This motivates the problem of few-shot IRL with multi-task demonstrations (FM-IRL), where an agent must learn a new task with sub"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.17760","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.17760/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.17760","created_at":"2026-07-21T02:21:58.779852+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.17760v1","created_at":"2026-07-21T02:21:58.779852+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.17760","created_at":"2026-07-21T02:21:58.779852+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWQVOG44RPEQ","created_at":"2026-07-21T02:21:58.779852+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWQVOG44RPEQVWTV","created_at":"2026-07-21T02:21:58.779852+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWQVOG44","created_at":"2026-07-21T02:21:58.779852+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO","json":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO.json","graph_json":"https://pith.science/api/pith-number/TWQVOG44RPEQVWTVRC5TGG6IIO/graph.json","events_json":"https://pith.science/api/pith-number/TWQVOG44RPEQVWTVRC5TGG6IIO/events.json","paper":"https://pith.science/paper/TWQVOG44"},"agent_actions":{"view_html":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO","download_json":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO.json","view_paper":"https://pith.science/paper/TWQVOG44","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.17760&json=true","fetch_graph":"https://pith.science/api/pith-number/TWQVOG44RPEQVWTVRC5TGG6IIO/graph.json","fetch_events":"https://pith.science/api/pith-number/TWQVOG44RPEQVWTVRC5TGG6IIO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO/action/storage_attestation","attest_author":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO/action/author_attestation","sign_citation":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO/action/citation_signature","submit_replication":"https://pith.science/pith/TWQVOG44RPEQVWTVRC5TGG6IIO/action/replication_record"}},"created_at":"2026-07-21T02:21:58.779852+00:00","updated_at":"2026-07-21T02:21:58.779852+00:00"}