{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V4GDBS37OIBIYEIDVS7QYKXDRG","short_pith_number":"pith:V4GDBS37","schema_version":"1.0","canonical_sha256":"af0c30cb7f72028c1103acbf0c2ae389b97295f5884add3d8871d6e7674ba11f","source":{"kind":"arxiv","id":"2402.13037","version":2},"attestation_state":"computed","paper":{"title":"Align Your Intents: Offline Imitation Learning via Optimal Transport","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dmitrii Krylov, Dmitry V. Dylov, Maksim Bobrin, Nazar Buzun","submitted_at":"2024-02-20T14:24:00Z","abstract_excerpt":"Offline Reinforcement Learning (RL) addresses the problem of sequential decision-making by learning optimal policy through pre-collected data, without interacting with the environment. As yet, it has remained somewhat impractical, because one rarely knows the reward explicitly and it is hard to distill it retrospectively. Here, we show that an imitating agent can still learn the desired behavior merely from observing the expert, despite the absence of explicit rewards or action labels. In our method, AILOT (Aligned Imitation Learning via Optimal Transport), we involve special representation of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13037","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-20T14:24:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"63956d89298771e9d33406968e6073a4df33e4793c961cc5816efdcab69cb418","abstract_canon_sha256":"057d87abf3fa2265c9f4bcf811e66e17fda32bed7a8b245c14b3616f48fa0632"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:29.976697Z","signature_b64":"6wRbzCw4vx3wdtzhuKdsUNiQZV2IWELAXrQ+YSq3Tq+XkKTT5PTgQTmi88yShnf7T0f6lNDx2NFnmKZm1vGTAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af0c30cb7f72028c1103acbf0c2ae389b97295f5884add3d8871d6e7674ba11f","last_reissued_at":"2026-07-05T09:15:29.976138Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:29.976138Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Align Your Intents: Offline Imitation Learning via Optimal Transport","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dmitrii Krylov, Dmitry V. Dylov, Maksim Bobrin, Nazar Buzun","submitted_at":"2024-02-20T14:24:00Z","abstract_excerpt":"Offline Reinforcement Learning (RL) addresses the problem of sequential decision-making by learning optimal policy through pre-collected data, without interacting with the environment. As yet, it has remained somewhat impractical, because one rarely knows the reward explicitly and it is hard to distill it retrospectively. Here, we show that an imitating agent can still learn the desired behavior merely from observing the expert, despite the absence of explicit rewards or action labels. In our method, AILOT (Aligned Imitation Learning via Optimal Transport), we involve special representation of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13037","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13037/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13037","created_at":"2026-07-05T09:15:29.976191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13037v2","created_at":"2026-07-05T09:15:29.976191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13037","created_at":"2026-07-05T09:15:29.976191+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4GDBS37OIBI","created_at":"2026-07-05T09:15:29.976191+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4GDBS37OIBIYEID","created_at":"2026-07-05T09:15:29.976191+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4GDBS37","created_at":"2026-07-05T09:15:29.976191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06793","citing_title":"Minimal Ingredients for Reward Assignment from Expert Demonstrations","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG","json":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG.json","graph_json":"https://pith.science/api/pith-number/V4GDBS37OIBIYEIDVS7QYKXDRG/graph.json","events_json":"https://pith.science/api/pith-number/V4GDBS37OIBIYEIDVS7QYKXDRG/events.json","paper":"https://pith.science/paper/V4GDBS37"},"agent_actions":{"view_html":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG","download_json":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG.json","view_paper":"https://pith.science/paper/V4GDBS37","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13037&json=true","fetch_graph":"https://pith.science/api/pith-number/V4GDBS37OIBIYEIDVS7QYKXDRG/graph.json","fetch_events":"https://pith.science/api/pith-number/V4GDBS37OIBIYEIDVS7QYKXDRG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG/action/storage_attestation","attest_author":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG/action/author_attestation","sign_citation":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG/action/citation_signature","submit_replication":"https://pith.science/pith/V4GDBS37OIBIYEIDVS7QYKXDRG/action/replication_record"}},"created_at":"2026-07-05T09:15:29.976191+00:00","updated_at":"2026-07-05T09:15:29.976191+00:00"}