{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NL2PHCU4IFOKH6OT4AQGQLACWY","short_pith_number":"pith:NL2PHCU4","schema_version":"1.0","canonical_sha256":"6af4f38a9c415ca3f9d3e020682c02b62e650adc4f6a5894aa62bfdf8bfc215b","source":{"kind":"arxiv","id":"2407.10967","version":2},"attestation_state":"computed","paper":{"title":"BECAUSE: Bilinear Causal Representation for Generalizable Offline Model-based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bo Li, Ding Zhao, Haohong Lin, Jiacheng Zhu, Jian Chen, Laixi Shi, Wenhao Ding","submitted_at":"2024-07-15T17:59:23Z","abstract_excerpt":"Offline model-based reinforcement learning (MBRL) enhances data efficiency by utilizing pre-collected datasets to learn models and policies, especially in scenarios where exploration is costly or infeasible. Nevertheless, its performance often suffers from the objective mismatch between model and policy learning, resulting in inferior performance despite accurate model predictions. This paper first identifies the primary source of this mismatch comes from the underlying confounders present in offline data for MBRL. Subsequently, we introduce \\textbf{B}ilin\\textbf{E}ar \\textbf{CAUS}al r\\textbf{"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10967","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-15T17:59:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3e3454e202238541d3a3c847f394027b2db5c10efb31dbb1c7921c350235f544","abstract_canon_sha256":"c69503c872f3a1340833fae2b8da3329a9775765980ebe6eb188d6d6caad01c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:02.167087Z","signature_b64":"BZqOdNy6gdMzq9OzeOqj+PY4LdUnKSnVywyIfs+gtaqok7GNkAfbAFNaDH81n3Y/vRdmMBSa57u9CJBaEd39DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6af4f38a9c415ca3f9d3e020682c02b62e650adc4f6a5894aa62bfdf8bfc215b","last_reissued_at":"2026-07-05T10:22:02.166564Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:02.166564Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BECAUSE: Bilinear Causal Representation for Generalizable Offline Model-based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bo Li, Ding Zhao, Haohong Lin, Jiacheng Zhu, Jian Chen, Laixi Shi, Wenhao Ding","submitted_at":"2024-07-15T17:59:23Z","abstract_excerpt":"Offline model-based reinforcement learning (MBRL) enhances data efficiency by utilizing pre-collected datasets to learn models and policies, especially in scenarios where exploration is costly or infeasible. Nevertheless, its performance often suffers from the objective mismatch between model and policy learning, resulting in inferior performance despite accurate model predictions. This paper first identifies the primary source of this mismatch comes from the underlying confounders present in offline data for MBRL. Subsequently, we introduce \\textbf{B}ilin\\textbf{E}ar \\textbf{CAUS}al r\\textbf{"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10967","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10967/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10967","created_at":"2026-07-05T10:22:02.166633+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10967v2","created_at":"2026-07-05T10:22:02.166633+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10967","created_at":"2026-07-05T10:22:02.166633+00:00"},{"alias_kind":"pith_short_12","alias_value":"NL2PHCU4IFOK","created_at":"2026-07-05T10:22:02.166633+00:00"},{"alias_kind":"pith_short_16","alias_value":"NL2PHCU4IFOKH6OT","created_at":"2026-07-05T10:22:02.166633+00:00"},{"alias_kind":"pith_short_8","alias_value":"NL2PHCU4","created_at":"2026-07-05T10:22:02.166633+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY","json":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY.json","graph_json":"https://pith.science/api/pith-number/NL2PHCU4IFOKH6OT4AQGQLACWY/graph.json","events_json":"https://pith.science/api/pith-number/NL2PHCU4IFOKH6OT4AQGQLACWY/events.json","paper":"https://pith.science/paper/NL2PHCU4"},"agent_actions":{"view_html":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY","download_json":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY.json","view_paper":"https://pith.science/paper/NL2PHCU4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10967&json=true","fetch_graph":"https://pith.science/api/pith-number/NL2PHCU4IFOKH6OT4AQGQLACWY/graph.json","fetch_events":"https://pith.science/api/pith-number/NL2PHCU4IFOKH6OT4AQGQLACWY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY/action/storage_attestation","attest_author":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY/action/author_attestation","sign_citation":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY/action/citation_signature","submit_replication":"https://pith.science/pith/NL2PHCU4IFOKH6OT4AQGQLACWY/action/replication_record"}},"created_at":"2026-07-05T10:22:02.166633+00:00","updated_at":"2026-07-05T10:22:02.166633+00:00"}