{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SJ6RYI747LNY4FG2NKQWI6EKUA","short_pith_number":"pith:SJ6RYI74","schema_version":"1.0","canonical_sha256":"927d1c23fcfadb8e14da6aa164788aa01ca0ba1eaaed2cea992389171229be81","source":{"kind":"arxiv","id":"2407.03967","version":2},"attestation_state":"computed","paper":{"title":"Investigating the Role of Instruction Variety and Task Difficulty in Robotic Manipulation Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CL","authors_text":"Alessandro Suglia, Amit Parekh, Ioannis Konstas, Nikolas Vitsakis","submitted_at":"2024-07-04T14:36:49Z","abstract_excerpt":"Evaluating the generalisation capabilities of multimodal models based solely on their performance on out-of-distribution data fails to capture their true robustness. This work introduces a comprehensive evaluation framework that systematically examines the role of instructions and inputs in the generalisation abilities of such models, considering architectural design, input perturbations across language and vision modalities, and increased task complexity. The proposed framework uncovers the resilience of multimodal models to extreme instruction perturbations and their vulnerability to observa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.03967","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-04T14:36:49Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"330655c26bd5bd522dbb46b837f7343b5bb158d50da755dbccbd77b1148c6999","abstract_canon_sha256":"1706e2286d9629256803f1c800b6d046b2646ceb50f14759edb6cfd12b836491"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:02.910905Z","signature_b64":"1EIDV/b8hPFs3ZVU3Vm0hdyioEhiTMOXTBxKmCUt1r/G7a7Vonifj1jo5bEXn2jcu7iCpgO1cIrbDV+yF50iDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"927d1c23fcfadb8e14da6aa164788aa01ca0ba1eaaed2cea992389171229be81","last_reissued_at":"2026-07-05T09:27:02.910435Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:02.910435Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating the Role of Instruction Variety and Task Difficulty in Robotic Manipulation Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CL","authors_text":"Alessandro Suglia, Amit Parekh, Ioannis Konstas, Nikolas Vitsakis","submitted_at":"2024-07-04T14:36:49Z","abstract_excerpt":"Evaluating the generalisation capabilities of multimodal models based solely on their performance on out-of-distribution data fails to capture their true robustness. This work introduces a comprehensive evaluation framework that systematically examines the role of instructions and inputs in the generalisation abilities of such models, considering architectural design, input perturbations across language and vision modalities, and increased task complexity. The proposed framework uncovers the resilience of multimodal models to extreme instruction perturbations and their vulnerability to observa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.03967","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.03967/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.03967","created_at":"2026-07-05T09:27:02.910501+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.03967v2","created_at":"2026-07-05T09:27:02.910501+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.03967","created_at":"2026-07-05T09:27:02.910501+00:00"},{"alias_kind":"pith_short_12","alias_value":"SJ6RYI747LNY","created_at":"2026-07-05T09:27:02.910501+00:00"},{"alias_kind":"pith_short_16","alias_value":"SJ6RYI747LNY4FG2","created_at":"2026-07-05T09:27:02.910501+00:00"},{"alias_kind":"pith_short_8","alias_value":"SJ6RYI74","created_at":"2026-07-05T09:27:02.910501+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.04246","citing_title":"SAFECAST: Robust Failure Detection for VLA Policies with Contrast-Set Training and Calibration","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA","json":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA.json","graph_json":"https://pith.science/api/pith-number/SJ6RYI747LNY4FG2NKQWI6EKUA/graph.json","events_json":"https://pith.science/api/pith-number/SJ6RYI747LNY4FG2NKQWI6EKUA/events.json","paper":"https://pith.science/paper/SJ6RYI74"},"agent_actions":{"view_html":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA","download_json":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA.json","view_paper":"https://pith.science/paper/SJ6RYI74","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.03967&json=true","fetch_graph":"https://pith.science/api/pith-number/SJ6RYI747LNY4FG2NKQWI6EKUA/graph.json","fetch_events":"https://pith.science/api/pith-number/SJ6RYI747LNY4FG2NKQWI6EKUA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA/action/storage_attestation","attest_author":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA/action/author_attestation","sign_citation":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA/action/citation_signature","submit_replication":"https://pith.science/pith/SJ6RYI747LNY4FG2NKQWI6EKUA/action/replication_record"}},"created_at":"2026-07-05T09:27:02.910501+00:00","updated_at":"2026-07-05T09:27:02.910501+00:00"}