{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OV5EJIKQH6ZR7CKFEBZLOUMK4X","short_pith_number":"pith:OV5EJIKQ","schema_version":"1.0","canonical_sha256":"757a44a1503fb31f89452072b7518ae5fc7ededd4badc8f700c9c36dd1f81b72","source":{"kind":"arxiv","id":"2401.17981","version":3},"attestation_state":"computed","paper":{"title":"From Training-Free to Adaptive: Empirical Insights into MLLMs' Understanding of Detection Information","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Daoyuan Chen, Qirui Jiao, Yaliang Li, Yilun Huang, Ying Shen","submitted_at":"2024-01-31T16:38:32Z","abstract_excerpt":"Despite the impressive capabilities of Multimodal Large Language Models (MLLMs) in integrating text and image modalities, challenges remain in accurately interpreting detailed visual elements. Vision detection models excel at recognizing fine-grained image details, prompting researchers to use them to enhance MLLMs. One effective strategy is to infuse detection information in text format, which has proven simple and effective. However, most studies utilize this method without training, leaving the potential of adaptive training largely unexplored. Adaptive training could significantly enhance "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.17981","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-31T16:38:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d42975a3df961aee9786cf4034d20cb87c1e5b22de4a4ef18cb25c4228f9c4f5","abstract_canon_sha256":"1a2ab4129ed874109f7f621d93f3fc09bbbcec130878a947646594960cab988a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:30.584961Z","signature_b64":"1HycsCAzLwkjufUXMJtipe85Jl0Zx1ZSlVUTwuDm6BCp26oVbyZCmTAwyB2bR4Gpaf1bpkIPOm29hogzknfoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"757a44a1503fb31f89452072b7518ae5fc7ededd4badc8f700c9c36dd1f81b72","last_reissued_at":"2026-07-05T09:51:30.584287Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:30.584287Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Training-Free to Adaptive: Empirical Insights into MLLMs' Understanding of Detection Information","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Daoyuan Chen, Qirui Jiao, Yaliang Li, Yilun Huang, Ying Shen","submitted_at":"2024-01-31T16:38:32Z","abstract_excerpt":"Despite the impressive capabilities of Multimodal Large Language Models (MLLMs) in integrating text and image modalities, challenges remain in accurately interpreting detailed visual elements. Vision detection models excel at recognizing fine-grained image details, prompting researchers to use them to enhance MLLMs. One effective strategy is to infuse detection information in text format, which has proven simple and effective. However, most studies utilize this method without training, leaving the potential of adaptive training largely unexplored. Adaptive training could significantly enhance "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.17981","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.17981/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.17981","created_at":"2026-07-05T09:51:30.584391+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.17981v3","created_at":"2026-07-05T09:51:30.584391+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.17981","created_at":"2026-07-05T09:51:30.584391+00:00"},{"alias_kind":"pith_short_12","alias_value":"OV5EJIKQH6ZR","created_at":"2026-07-05T09:51:30.584391+00:00"},{"alias_kind":"pith_short_16","alias_value":"OV5EJIKQH6ZR7CKF","created_at":"2026-07-05T09:51:30.584391+00:00"},{"alias_kind":"pith_short_8","alias_value":"OV5EJIKQ","created_at":"2026-07-05T09:51:30.584391+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.12844","citing_title":"GuideDog: A Real-World Egocentric Multimodal Dataset for Blind and Low-Vision Accessibility-Aware Guidance","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01359","citing_title":"Structural Ranking of the Cognitive Plausibility of Computational Models of Analogy and Metaphors with the Minimal Cognitive Grid","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":81,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X","json":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X.json","graph_json":"https://pith.science/api/pith-number/OV5EJIKQH6ZR7CKFEBZLOUMK4X/graph.json","events_json":"https://pith.science/api/pith-number/OV5EJIKQH6ZR7CKFEBZLOUMK4X/events.json","paper":"https://pith.science/paper/OV5EJIKQ"},"agent_actions":{"view_html":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X","download_json":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X.json","view_paper":"https://pith.science/paper/OV5EJIKQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.17981&json=true","fetch_graph":"https://pith.science/api/pith-number/OV5EJIKQH6ZR7CKFEBZLOUMK4X/graph.json","fetch_events":"https://pith.science/api/pith-number/OV5EJIKQH6ZR7CKFEBZLOUMK4X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X/action/storage_attestation","attest_author":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X/action/author_attestation","sign_citation":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X/action/citation_signature","submit_replication":"https://pith.science/pith/OV5EJIKQH6ZR7CKFEBZLOUMK4X/action/replication_record"}},"created_at":"2026-07-05T09:51:30.584391+00:00","updated_at":"2026-07-05T09:51:30.584391+00:00"}