{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IWZK2MBOGHSKUL2P33AR73DZ3Z","short_pith_number":"pith:IWZK2MBO","schema_version":"1.0","canonical_sha256":"45b2ad302e31e4aa2f4fdec11fec79de66c7da3d2cbdbdc56a4858c38a3050c2","source":{"kind":"arxiv","id":"2306.03792","version":3},"attestation_state":"computed","paper":{"title":"FAMO: Fast Adaptive Multitask Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Liu, Peter Stone, Qiang Liu, Yihao Feng","submitted_at":"2023-06-06T15:39:54Z","abstract_excerpt":"One of the grand enduring goals of AI is to create generalist agents that can learn multiple different tasks from diverse data via multitask learning (MTL). However, in practice, applying gradient descent (GD) on the average loss across all tasks may yield poor multitask performance due to severe under-optimization of certain tasks. Previous approaches that manipulate task gradients for a more balanced loss decrease require storing and computing all task gradients ($\\mathcal{O}(k)$ space and time where $k$ is the number of tasks), limiting their use in large-scale scenarios. In this work, we i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.03792","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-06T15:39:54Z","cross_cats_sorted":[],"title_canon_sha256":"17af162d4a2cad779c39e23a56e906090475ecc1d8510c796f1c8c639b67e1ea","abstract_canon_sha256":"d9fe033c24e5d2bf2209b08d14d9e4863602db05ddde220957d8847ef603263f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:29.439397Z","signature_b64":"b0c3cyCjZwP+AHiDIBcZIRykJSPfkIm5ec/p1jXa9j+H7ClGjAeQK1n/Ott4YHuEaldxk+iFR8SQwy1f8R+WAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45b2ad302e31e4aa2f4fdec11fec79de66c7da3d2cbdbdc56a4858c38a3050c2","last_reissued_at":"2026-07-05T07:06:29.438895Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:29.438895Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FAMO: Fast Adaptive Multitask Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Liu, Peter Stone, Qiang Liu, Yihao Feng","submitted_at":"2023-06-06T15:39:54Z","abstract_excerpt":"One of the grand enduring goals of AI is to create generalist agents that can learn multiple different tasks from diverse data via multitask learning (MTL). However, in practice, applying gradient descent (GD) on the average loss across all tasks may yield poor multitask performance due to severe under-optimization of certain tasks. Previous approaches that manipulate task gradients for a more balanced loss decrease require storing and computing all task gradients ($\\mathcal{O}(k)$ space and time where $k$ is the number of tasks), limiting their use in large-scale scenarios. In this work, we i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.03792","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.03792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.03792","created_at":"2026-07-05T07:06:29.438952+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.03792v3","created_at":"2026-07-05T07:06:29.438952+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.03792","created_at":"2026-07-05T07:06:29.438952+00:00"},{"alias_kind":"pith_short_12","alias_value":"IWZK2MBOGHSK","created_at":"2026-07-05T07:06:29.438952+00:00"},{"alias_kind":"pith_short_16","alias_value":"IWZK2MBOGHSKUL2P","created_at":"2026-07-05T07:06:29.438952+00:00"},{"alias_kind":"pith_short_8","alias_value":"IWZK2MBO","created_at":"2026-07-05T07:06:29.438952+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00078","citing_title":"Exploring Line Bundle Standard Models with Transformers","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03335","citing_title":"GPU-Parallel Multi-Task Reinforcement Learning with Demonstration Guided Policy Optimization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11473","citing_title":"TOPPO: Rethinking PPO for Multi-Task Reinforcement Learning with Critic Balancing","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z","json":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z.json","graph_json":"https://pith.science/api/pith-number/IWZK2MBOGHSKUL2P33AR73DZ3Z/graph.json","events_json":"https://pith.science/api/pith-number/IWZK2MBOGHSKUL2P33AR73DZ3Z/events.json","paper":"https://pith.science/paper/IWZK2MBO"},"agent_actions":{"view_html":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z","download_json":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z.json","view_paper":"https://pith.science/paper/IWZK2MBO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.03792&json=true","fetch_graph":"https://pith.science/api/pith-number/IWZK2MBOGHSKUL2P33AR73DZ3Z/graph.json","fetch_events":"https://pith.science/api/pith-number/IWZK2MBOGHSKUL2P33AR73DZ3Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z/action/storage_attestation","attest_author":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z/action/author_attestation","sign_citation":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z/action/citation_signature","submit_replication":"https://pith.science/pith/IWZK2MBOGHSKUL2P33AR73DZ3Z/action/replication_record"}},"created_at":"2026-07-05T07:06:29.438952+00:00","updated_at":"2026-07-05T07:06:29.438952+00:00"}