{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FT66N5C5VAZT5PO4QUTFQYBHZJ","short_pith_number":"pith:FT66N5C5","schema_version":"1.0","canonical_sha256":"2cfde6f45da8333ebddc8526586027ca741ebd527efc6dfc4fb43571103cd5db","source":{"kind":"arxiv","id":"2404.08506","version":1},"attestation_state":"computed","paper":{"title":"LaSagnA: Language-based Segmentation Assistant for Complex Queries","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wei, Haoxian Tan, Lin Ma, Yujie Zhong, Yujiu Yang","submitted_at":"2024-04-12T14:40:45Z","abstract_excerpt":"Recent advancements have empowered Large Language Models for Vision (vLLMs) to generate detailed perceptual outcomes, including bounding boxes and masks. Nonetheless, there are two constraints that restrict the further application of these vLLMs: the incapability of handling multiple targets per query and the failure to identify the absence of query objects in the image. In this study, we acknowledge that the main cause of these problems is the insufficient complexity of training queries. Consequently, we define the general sequence format for complex queries. Then we incorporate a semantic se"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.08506","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-12T14:40:45Z","cross_cats_sorted":[],"title_canon_sha256":"8b4acb6ecb345161b564c77a00f53e032dda9cd562780fc55d613aaca423390d","abstract_canon_sha256":"f246e353169d314525d42617befbf15b51a1182fe80f8d249d0db584b78a7721"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:22.818602Z","signature_b64":"s+r1LaHVuyQF4MSHfwWb/qtY75WhopA4QC9ziSVoA+3rSReVek3vcrkgNGA9Uv13oq5pTtsgJyaTO0foVbnMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cfde6f45da8333ebddc8526586027ca741ebd527efc6dfc4fb43571103cd5db","last_reissued_at":"2026-07-05T08:07:22.818150Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:22.818150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LaSagnA: Language-based Segmentation Assistant for Complex Queries","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wei, Haoxian Tan, Lin Ma, Yujie Zhong, Yujiu Yang","submitted_at":"2024-04-12T14:40:45Z","abstract_excerpt":"Recent advancements have empowered Large Language Models for Vision (vLLMs) to generate detailed perceptual outcomes, including bounding boxes and masks. Nonetheless, there are two constraints that restrict the further application of these vLLMs: the incapability of handling multiple targets per query and the failure to identify the absence of query objects in the image. In this study, we acknowledge that the main cause of these problems is the insufficient complexity of training queries. Consequently, we define the general sequence format for complex queries. Then we incorporate a semantic se"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.08506","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.08506/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.08506","created_at":"2026-07-05T08:07:22.818207+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.08506v1","created_at":"2026-07-05T08:07:22.818207+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.08506","created_at":"2026-07-05T08:07:22.818207+00:00"},{"alias_kind":"pith_short_12","alias_value":"FT66N5C5VAZT","created_at":"2026-07-05T08:07:22.818207+00:00"},{"alias_kind":"pith_short_16","alias_value":"FT66N5C5VAZT5PO4","created_at":"2026-07-05T08:07:22.818207+00:00"},{"alias_kind":"pith_short_8","alias_value":"FT66N5C5","created_at":"2026-07-05T08:07:22.818207+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05798","citing_title":"Segmentation before Answering: Pixel Grounding for MLLM Visual Reasoning","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06760","citing_title":"MedSIGHT: Towards Grounded Visual Comprehension in Medical Large Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31924","citing_title":"InstanceControl: Controllable Complex Image Generation without Instance Labeling","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24553","citing_title":"IQA-Spider: Unifying Multi-Granularity Image Quality Assessment with Reasoning, Grounding and Referring","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29267","citing_title":"Enhancing Part-Level Point Grounding for Any Open-Source MLLMs","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20676","citing_title":"VISTAQA: Benchmarking Joint Visual Question Answering and Pixel-Level Evidence","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2501.04001","citing_title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00891","citing_title":"X2SAM: Any Segmentation in Images and Videos","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12315","citing_title":"GTPBD-MM: A Global Terraced Parcel and Boundary Dataset with Multi-Modality","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15670","citing_title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ","json":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ.json","graph_json":"https://pith.science/api/pith-number/FT66N5C5VAZT5PO4QUTFQYBHZJ/graph.json","events_json":"https://pith.science/api/pith-number/FT66N5C5VAZT5PO4QUTFQYBHZJ/events.json","paper":"https://pith.science/paper/FT66N5C5"},"agent_actions":{"view_html":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ","download_json":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ.json","view_paper":"https://pith.science/paper/FT66N5C5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.08506&json=true","fetch_graph":"https://pith.science/api/pith-number/FT66N5C5VAZT5PO4QUTFQYBHZJ/graph.json","fetch_events":"https://pith.science/api/pith-number/FT66N5C5VAZT5PO4QUTFQYBHZJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ/action/storage_attestation","attest_author":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ/action/author_attestation","sign_citation":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ/action/citation_signature","submit_replication":"https://pith.science/pith/FT66N5C5VAZT5PO4QUTFQYBHZJ/action/replication_record"}},"created_at":"2026-07-05T08:07:22.818207+00:00","updated_at":"2026-07-05T08:07:22.818207+00:00"}