{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5Y2V57JNVBOBYWPY2I4Y22ZSWO","short_pith_number":"pith:5Y2V57JN","schema_version":"1.0","canonical_sha256":"ee355efd2da85c1c59f8d2398d6b32b3a36ac26922b32d01d9ef7893c9ddaf5e","source":{"kind":"arxiv","id":"2307.13721","version":1},"attestation_state":"computed","paper":{"title":"Foundational Models Defining a New Era in Vision: A Survey and Outlook","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fahad Shahbaz Khan, Hisham Cholakkal, Ming-Hsuan Yang, Mubarak Shah, Muhammad Awais, Muzammal Naseer, Rao Muhammad Anwer, Salman Khan","submitted_at":"2023-07-25T17:59:18Z","abstract_excerpt":"Vision systems to see and reason about the compositional nature of visual scenes are fundamental to understanding our world. The complex relations between objects and their locations, ambiguities, and variations in the real-world environment can be better described in human language, naturally governed by grammatical rules and other modalities such as audio and depth. The models learned to bridge the gap between such modalities coupled with large-scale training data facilitate contextual reasoning, generalization, and prompt capabilities at test time. These models are referred to as foundation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.13721","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-07-25T17:59:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2b13e45c6087d56e070d61fe23a55bd70b103cea16bada74ba026f62cb5675d8","abstract_canon_sha256":"db98a9a733a78a9d299e81b2cdafb1b3962d91241bc4593b5b06ba52829bd92a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:34:51.662741Z","signature_b64":"FR6WJgrn2No0VGVLENEDvD1S2ru48PNtlMr6A+rZJ2sfPlEd8SgWrWVQf2HV3he1OPwSxec5HdvgVzeOa4IiAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee355efd2da85c1c59f8d2398d6b32b3a36ac26922b32d01d9ef7893c9ddaf5e","last_reissued_at":"2026-07-05T06:34:51.662265Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:34:51.662265Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Foundational Models Defining a New Era in Vision: A Survey and Outlook","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fahad Shahbaz Khan, Hisham Cholakkal, Ming-Hsuan Yang, Mubarak Shah, Muhammad Awais, Muzammal Naseer, Rao Muhammad Anwer, Salman Khan","submitted_at":"2023-07-25T17:59:18Z","abstract_excerpt":"Vision systems to see and reason about the compositional nature of visual scenes are fundamental to understanding our world. The complex relations between objects and their locations, ambiguities, and variations in the real-world environment can be better described in human language, naturally governed by grammatical rules and other modalities such as audio and depth. The models learned to bridge the gap between such modalities coupled with large-scale training data facilitate contextual reasoning, generalization, and prompt capabilities at test time. These models are referred to as foundation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.13721","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.13721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.13721","created_at":"2026-07-05T06:34:51.662336+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.13721v1","created_at":"2026-07-05T06:34:51.662336+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.13721","created_at":"2026-07-05T06:34:51.662336+00:00"},{"alias_kind":"pith_short_12","alias_value":"5Y2V57JNVBOB","created_at":"2026-07-05T06:34:51.662336+00:00"},{"alias_kind":"pith_short_16","alias_value":"5Y2V57JNVBOBYWPY","created_at":"2026-07-05T06:34:51.662336+00:00"},{"alias_kind":"pith_short_8","alias_value":"5Y2V57JN","created_at":"2026-07-05T06:34:51.662336+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03713","citing_title":"Investigating Adversarial Robustness of Multi-modal Large Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2408.00923","citing_title":"Reclaiming Residual Knowledge: A Novel Paradigm to Low-Bit Quantization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10440","citing_title":"LLaVA-CoT: Let Vision Language Models Reason Step-by-Step","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14994","citing_title":"Degradation-aware Predictive Energy Management for Fuel Cell-Battery Ship Power System with Data-driven Load Forecasting","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27351","citing_title":"Heterogeneous Scientific Foundation Model Collaboration","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO","json":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO.json","graph_json":"https://pith.science/api/pith-number/5Y2V57JNVBOBYWPY2I4Y22ZSWO/graph.json","events_json":"https://pith.science/api/pith-number/5Y2V57JNVBOBYWPY2I4Y22ZSWO/events.json","paper":"https://pith.science/paper/5Y2V57JN"},"agent_actions":{"view_html":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO","download_json":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO.json","view_paper":"https://pith.science/paper/5Y2V57JN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.13721&json=true","fetch_graph":"https://pith.science/api/pith-number/5Y2V57JNVBOBYWPY2I4Y22ZSWO/graph.json","fetch_events":"https://pith.science/api/pith-number/5Y2V57JNVBOBYWPY2I4Y22ZSWO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO/action/storage_attestation","attest_author":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO/action/author_attestation","sign_citation":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO/action/citation_signature","submit_replication":"https://pith.science/pith/5Y2V57JNVBOBYWPY2I4Y22ZSWO/action/replication_record"}},"created_at":"2026-07-05T06:34:51.662336+00:00","updated_at":"2026-07-05T06:34:51.662336+00:00"}