{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TKMT4HHTTCDOHTPOXLPWXAH4T2","short_pith_number":"pith:TKMT4HHT","schema_version":"1.0","canonical_sha256":"9a993e1cf39886e3cdeebadf6b80fc9ebfcfa91a3168a2b146c2a6801edd138d","source":{"kind":"arxiv","id":"2506.06793","version":1},"attestation_state":"computed","paper":{"title":"Is Optimal Transport Necessary for Inverse Reinforcement Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Keith Ross, Yumi Omori, Zixuan Dong","submitted_at":"2025-06-07T13:29:37Z","abstract_excerpt":"Inverse Reinforcement Learning (IRL) aims to recover a reward function from expert demonstrations. Recently, Optimal Transport (OT) methods have been successfully deployed to align trajectories and infer rewards. While OT-based methods have shown strong empirical results, they introduce algorithmic complexity, hyperparameter sensitivity, and require solving the OT optimization problems. In this work, we challenge the necessity of OT in IRL by proposing two simple, heuristic alternatives: (1) Minimum-Distance Reward, which assigns rewards based on the nearest expert state regardless of temporal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.06793","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-07T13:29:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fd8f5a6cce2d813f976d3afd5b91ca470aa05f497d7e30f046c7a0db08c0879a","abstract_canon_sha256":"d929b06edc5c44ad3b8fd8a19e42b119489fd7bab083a430ce74cd5c65fae201"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:55.722256Z","signature_b64":"QauZ1KQklXg3vGKnv1rR9Slweqzs9IEFRjWeNdawzNHqDxpuogM7WLgoQ9XstNefusEREStLhj9GRm+/0i9CDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a993e1cf39886e3cdeebadf6b80fc9ebfcfa91a3168a2b146c2a6801edd138d","last_reissued_at":"2026-07-05T11:17:55.721876Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:55.721876Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Optimal Transport Necessary for Inverse Reinforcement Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Keith Ross, Yumi Omori, Zixuan Dong","submitted_at":"2025-06-07T13:29:37Z","abstract_excerpt":"Inverse Reinforcement Learning (IRL) aims to recover a reward function from expert demonstrations. Recently, Optimal Transport (OT) methods have been successfully deployed to align trajectories and infer rewards. While OT-based methods have shown strong empirical results, they introduce algorithmic complexity, hyperparameter sensitivity, and require solving the OT optimization problems. In this work, we challenge the necessity of OT in IRL by proposing two simple, heuristic alternatives: (1) Minimum-Distance Reward, which assigns rewards based on the nearest expert state regardless of temporal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.06793","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.06793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.06793","created_at":"2026-07-05T11:17:55.721930+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.06793v1","created_at":"2026-07-05T11:17:55.721930+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.06793","created_at":"2026-07-05T11:17:55.721930+00:00"},{"alias_kind":"pith_short_12","alias_value":"TKMT4HHTTCDO","created_at":"2026-07-05T11:17:55.721930+00:00"},{"alias_kind":"pith_short_16","alias_value":"TKMT4HHTTCDOHTPO","created_at":"2026-07-05T11:17:55.721930+00:00"},{"alias_kind":"pith_short_8","alias_value":"TKMT4HHT","created_at":"2026-07-05T11:17:55.721930+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2","json":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2.json","graph_json":"https://pith.science/api/pith-number/TKMT4HHTTCDOHTPOXLPWXAH4T2/graph.json","events_json":"https://pith.science/api/pith-number/TKMT4HHTTCDOHTPOXLPWXAH4T2/events.json","paper":"https://pith.science/paper/TKMT4HHT"},"agent_actions":{"view_html":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2","download_json":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2.json","view_paper":"https://pith.science/paper/TKMT4HHT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.06793&json=true","fetch_graph":"https://pith.science/api/pith-number/TKMT4HHTTCDOHTPOXLPWXAH4T2/graph.json","fetch_events":"https://pith.science/api/pith-number/TKMT4HHTTCDOHTPOXLPWXAH4T2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2/action/storage_attestation","attest_author":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2/action/author_attestation","sign_citation":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2/action/citation_signature","submit_replication":"https://pith.science/pith/TKMT4HHTTCDOHTPOXLPWXAH4T2/action/replication_record"}},"created_at":"2026-07-05T11:17:55.721930+00:00","updated_at":"2026-07-05T11:17:55.721930+00:00"}