{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YDBPSDPIJDB2XQYQ7G7GX2YIZJ","short_pith_number":"pith:YDBPSDPI","schema_version":"1.0","canonical_sha256":"c0c2f90de848c3abc310f9be6beb08ca7c9ace7eedce8d24e099dbb961fc4692","source":{"kind":"arxiv","id":"2408.03160","version":2},"attestation_state":"computed","paper":{"title":"User-in-the-loop Evaluation of Multimodal LLMs for Activity Assistance","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Brian Chen, Hamid Eghbalzadeh, Mrinal Verghese, Ruta Desai, Tushar Nagarajan","submitted_at":"2024-08-04T06:12:42Z","abstract_excerpt":"Our research investigates the capability of modern multimodal reasoning models, powered by Large Language Models (LLMs), to facilitate vision-powered assistants for multi-step daily activities. Such assistants must be able to 1) encode relevant visual history from the assistant's sensors, e.g., camera, 2) forecast future actions for accomplishing the activity, and 3) replan based on the user in the loop. To evaluate the first two capabilities, grounding visual history and forecasting in short and long horizons, we conduct benchmarking of two prominent classes of multimodal LLM approaches -- So"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.03160","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-04T06:12:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bb6ca1152c3e4a7c0375de34af1ed51de962541b4fb3f86fb325ab981303596d","abstract_canon_sha256":"7eca5554f972a1a463d4a6f084febfb38d45be8c375993a377579e8696f9b3cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:55.075254Z","signature_b64":"D/E+Hxc8nxdSlbBrlCZYejbMlq5ABBYPVLp5SavEzcUwhB771WRDoH2QGK8TIT3+HJx8viM1k0EieJ1nxUnlAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0c2f90de848c3abc310f9be6beb08ca7c9ace7eedce8d24e099dbb961fc4692","last_reissued_at":"2026-07-05T08:54:55.074809Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:55.074809Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"User-in-the-loop Evaluation of Multimodal LLMs for Activity Assistance","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Brian Chen, Hamid Eghbalzadeh, Mrinal Verghese, Ruta Desai, Tushar Nagarajan","submitted_at":"2024-08-04T06:12:42Z","abstract_excerpt":"Our research investigates the capability of modern multimodal reasoning models, powered by Large Language Models (LLMs), to facilitate vision-powered assistants for multi-step daily activities. Such assistants must be able to 1) encode relevant visual history from the assistant's sensors, e.g., camera, 2) forecast future actions for accomplishing the activity, and 3) replan based on the user in the loop. To evaluate the first two capabilities, grounding visual history and forecasting in short and long horizons, we conduct benchmarking of two prominent classes of multimodal LLM approaches -- So"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.03160","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.03160/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.03160","created_at":"2026-07-05T08:54:55.074872+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.03160v2","created_at":"2026-07-05T08:54:55.074872+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.03160","created_at":"2026-07-05T08:54:55.074872+00:00"},{"alias_kind":"pith_short_12","alias_value":"YDBPSDPIJDB2","created_at":"2026-07-05T08:54:55.074872+00:00"},{"alias_kind":"pith_short_16","alias_value":"YDBPSDPIJDB2XQYQ","created_at":"2026-07-05T08:54:55.074872+00:00"},{"alias_kind":"pith_short_8","alias_value":"YDBPSDPI","created_at":"2026-07-05T08:54:55.074872+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ","json":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ.json","graph_json":"https://pith.science/api/pith-number/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/graph.json","events_json":"https://pith.science/api/pith-number/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/events.json","paper":"https://pith.science/paper/YDBPSDPI"},"agent_actions":{"view_html":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ","download_json":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ.json","view_paper":"https://pith.science/paper/YDBPSDPI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.03160&json=true","fetch_graph":"https://pith.science/api/pith-number/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/graph.json","fetch_events":"https://pith.science/api/pith-number/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/action/storage_attestation","attest_author":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/action/author_attestation","sign_citation":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/action/citation_signature","submit_replication":"https://pith.science/pith/YDBPSDPIJDB2XQYQ7G7GX2YIZJ/action/replication_record"}},"created_at":"2026-07-05T08:54:55.074872+00:00","updated_at":"2026-07-05T08:54:55.074872+00:00"}