{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:67NJS2H4W5DCSVFLC3CGJNJIWA","short_pith_number":"pith:67NJS2H4","schema_version":"1.0","canonical_sha256":"f7da9968fcb7462954ab16c464b528b03ae03527e4513581ba7c0b5b07744be1","source":{"kind":"arxiv","id":"2310.00166","version":1},"attestation_state":"computed","paper":{"title":"Motif: Intrinsic Motivation from Artificial Intelligence Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Amy Zhang, Martin Klissarov, Mikael Henaff, Pascal Vincent, Pierluca D'Oro, Pierre-Luc Bacon, Roberta Raileanu, Shagun Sodhani","submitted_at":"2023-09-29T22:10:01Z","abstract_excerpt":"Exploring rich environments and evaluating one's actions without prior knowledge is immensely challenging. In this paper, we propose Motif, a general method to interface such prior knowledge from a Large Language Model (LLM) with an agent. Motif is based on the idea of grounding LLMs for decision-making without requiring them to interact with the environment: it elicits preferences from an LLM over pairs of captions to construct an intrinsic reward, which is then used to train agents with reinforcement learning. We evaluate Motif's performance and behavior on the challenging, open-ended and pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.00166","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-29T22:10:01Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"73b3323fb1e44ba72fcea34eee95996ef87ebf003ff243b8b15e45d804b147b8","abstract_canon_sha256":"5842e2251debb563856cae632f6bccdeba23bb00b0c875e112746416f71e17ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:56:02.181512Z","signature_b64":"iT5zuHy940kB3itPZOdfaT16UvpWhMfyd7/8NdyZ2fvTVjeZdZ9sb8B7w4lC25wV2ikE+plRA5MPpQMNW0zjCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7da9968fcb7462954ab16c464b528b03ae03527e4513581ba7c0b5b07744be1","last_reissued_at":"2026-07-05T06:56:02.180969Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:56:02.180969Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Motif: Intrinsic Motivation from Artificial Intelligence Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Amy Zhang, Martin Klissarov, Mikael Henaff, Pascal Vincent, Pierluca D'Oro, Pierre-Luc Bacon, Roberta Raileanu, Shagun Sodhani","submitted_at":"2023-09-29T22:10:01Z","abstract_excerpt":"Exploring rich environments and evaluating one's actions without prior knowledge is immensely challenging. In this paper, we propose Motif, a general method to interface such prior knowledge from a Large Language Model (LLM) with an agent. Motif is based on the idea of grounding LLMs for decision-making without requiring them to interact with the environment: it elicits preferences from an LLM over pairs of captions to construct an intrinsic reward, which is then used to train agents with reinforcement learning. We evaluate Motif's performance and behavior on the challenging, open-ended and pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00166","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00166/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.00166","created_at":"2026-07-05T06:56:02.181029+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.00166v1","created_at":"2026-07-05T06:56:02.181029+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00166","created_at":"2026-07-05T06:56:02.181029+00:00"},{"alias_kind":"pith_short_12","alias_value":"67NJS2H4W5DC","created_at":"2026-07-05T06:56:02.181029+00:00"},{"alias_kind":"pith_short_16","alias_value":"67NJS2H4W5DCSVFL","created_at":"2026-07-05T06:56:02.181029+00:00"},{"alias_kind":"pith_short_8","alias_value":"67NJS2H4","created_at":"2026-07-05T06:56:02.181029+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00483","citing_title":"VLM-AR3L: Vision-Language Models for Absolute and Relative Rewards in Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00483","citing_title":"VLM-AR3L: Vision-Language Models for Absolute and Relative Rewards in Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23551","citing_title":"Goal-Conditioned Agents that Learn Everything All at Once","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24558","citing_title":"Hierarchical Behaviour Spaces","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06869","citing_title":"Agentick: A Unified Benchmark for General Sequential Decision-Making Agents","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA","json":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA.json","graph_json":"https://pith.science/api/pith-number/67NJS2H4W5DCSVFLC3CGJNJIWA/graph.json","events_json":"https://pith.science/api/pith-number/67NJS2H4W5DCSVFLC3CGJNJIWA/events.json","paper":"https://pith.science/paper/67NJS2H4"},"agent_actions":{"view_html":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA","download_json":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA.json","view_paper":"https://pith.science/paper/67NJS2H4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.00166&json=true","fetch_graph":"https://pith.science/api/pith-number/67NJS2H4W5DCSVFLC3CGJNJIWA/graph.json","fetch_events":"https://pith.science/api/pith-number/67NJS2H4W5DCSVFLC3CGJNJIWA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA/action/storage_attestation","attest_author":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA/action/author_attestation","sign_citation":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA/action/citation_signature","submit_replication":"https://pith.science/pith/67NJS2H4W5DCSVFLC3CGJNJIWA/action/replication_record"}},"created_at":"2026-07-05T06:56:02.181029+00:00","updated_at":"2026-07-05T06:56:02.181029+00:00"}