{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LMD4MTEJO6ZQFR3S7RBXSS3KRY","short_pith_number":"pith:LMD4MTEJ","schema_version":"1.0","canonical_sha256":"5b07c64c8977b302c772fc43794b6a8e31ad4c913642ad363df320f1165ae8bb","source":{"kind":"arxiv","id":"2411.16761","version":2},"attestation_state":"computed","paper":{"title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bumsoo Kim, Buru Chang, Eun Tae Kim, Ji Hyeok Jung, Joo Ho Lee, Seoyeon Kim","submitted_at":"2024-11-24T15:07:47Z","abstract_excerpt":"Multimodal large language models (MLLMs) act as essential interfaces, connecting humans with AI technologies in multimodal applications. However, current MLLMs face challenges in accurately interpreting object orientation in images due to inconsistent orientation annotations in training data, hindering the development of a coherent orientation understanding. To overcome this, we propose egocentric instruction tuning, which aligns MLLMs' orientation understanding with the user's perspective, based on a consistent annotation standard derived from the user's egocentric viewpoint. We first generat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.16761","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-24T15:07:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"56d4540038d8d27c4127d3d6370e622bcbc3e91ce26f2058e1d318f4e031ca44","abstract_canon_sha256":"0697c6b2c21d0803f3fd952e696dd5df55cbeea8f5f2f783a6e4177027610972"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:17.602664Z","signature_b64":"hGJ7brjomdlkHHjF8EF0cL/u6Mjn4XB+WlOUq7Q9zvtDKeNgx9eXczifX0qT4CMWOHXVX0V0D7SiVQbyyNFRBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b07c64c8977b302c772fc43794b6a8e31ad4c913642ad363df320f1165ae8bb","last_reissued_at":"2026-07-05T10:41:17.602120Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:17.602120Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is 'Right' Right? Enhancing Object Orientation Understanding in Multimodal Large Language Models through Egocentric Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bumsoo Kim, Buru Chang, Eun Tae Kim, Ji Hyeok Jung, Joo Ho Lee, Seoyeon Kim","submitted_at":"2024-11-24T15:07:47Z","abstract_excerpt":"Multimodal large language models (MLLMs) act as essential interfaces, connecting humans with AI technologies in multimodal applications. However, current MLLMs face challenges in accurately interpreting object orientation in images due to inconsistent orientation annotations in training data, hindering the development of a coherent orientation understanding. To overcome this, we propose egocentric instruction tuning, which aligns MLLMs' orientation understanding with the user's perspective, based on a consistent annotation standard derived from the user's egocentric viewpoint. We first generat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.16761","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.16761/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.16761","created_at":"2026-07-05T10:41:17.602182+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.16761v2","created_at":"2026-07-05T10:41:17.602182+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.16761","created_at":"2026-07-05T10:41:17.602182+00:00"},{"alias_kind":"pith_short_12","alias_value":"LMD4MTEJO6ZQ","created_at":"2026-07-05T10:41:17.602182+00:00"},{"alias_kind":"pith_short_16","alias_value":"LMD4MTEJO6ZQFR3S","created_at":"2026-07-05T10:41:17.602182+00:00"},{"alias_kind":"pith_short_8","alias_value":"LMD4MTEJ","created_at":"2026-07-05T10:41:17.602182+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.09082","citing_title":"AVA-Bench: Atomic Visual Ability Benchmark for Vision Foundation Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04746","citing_title":"Think in Strokes, Not Pixels: Process-Driven Image Generation via Interleaved Reasoning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY","json":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY.json","graph_json":"https://pith.science/api/pith-number/LMD4MTEJO6ZQFR3S7RBXSS3KRY/graph.json","events_json":"https://pith.science/api/pith-number/LMD4MTEJO6ZQFR3S7RBXSS3KRY/events.json","paper":"https://pith.science/paper/LMD4MTEJ"},"agent_actions":{"view_html":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY","download_json":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY.json","view_paper":"https://pith.science/paper/LMD4MTEJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.16761&json=true","fetch_graph":"https://pith.science/api/pith-number/LMD4MTEJO6ZQFR3S7RBXSS3KRY/graph.json","fetch_events":"https://pith.science/api/pith-number/LMD4MTEJO6ZQFR3S7RBXSS3KRY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY/action/storage_attestation","attest_author":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY/action/author_attestation","sign_citation":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY/action/citation_signature","submit_replication":"https://pith.science/pith/LMD4MTEJO6ZQFR3S7RBXSS3KRY/action/replication_record"}},"created_at":"2026-07-05T10:41:17.602182+00:00","updated_at":"2026-07-05T10:41:17.602182+00:00"}