{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HUPC5NNAPCMNBQMIGWNGJIE4WH","short_pith_number":"pith:HUPC5NNA","schema_version":"1.0","canonical_sha256":"3d1e2eb5a07898d0c188359a64a09cb1fcf194d5ecbda44eee182619d3a1b63b","source":{"kind":"arxiv","id":"2410.15364","version":1},"attestation_state":"computed","paper":{"title":"Scene Graph Generation with Role-Playing Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Guikun Chen, Jin Li, Wenguan Wang","submitted_at":"2024-10-20T11:40:31Z","abstract_excerpt":"Current approaches for open-vocabulary scene graph generation (OVSGG) use vision-language models such as CLIP and follow a standard zero-shot pipeline -- computing similarity between the query image and the text embeddings for each category (i.e., text classifiers). In this work, we argue that the text classifiers adopted by existing OVSGG methods, i.e., category-/part-level prompts, are scene-agnostic as they remain unchanged across contexts. Using such fixed text classifiers not only struggles to model visual relations with high variance, but also falls short in adapting to distinct contexts"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15364","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-20T11:40:31Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"387d13f2189f43deeac9e19528385ac94fd96477894e2b4386e286ee1ff91b67","abstract_canon_sha256":"17fb5e75dda26af58c51e4f8887e336472f72c26917f2812efa3b147fd911e3e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:18.404099Z","signature_b64":"rbq7MJjd+mTz8yxgGnqPW5sgSJ3LO2xszzl7QFCNYadnqKSKw/jC+djzPcdsAJmN6c8FF7tlL4fe99GjYJeAAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d1e2eb5a07898d0c188359a64a09cb1fcf194d5ecbda44eee182619d3a1b63b","last_reissued_at":"2026-07-05T09:23:18.403686Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:18.403686Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scene Graph Generation with Role-Playing Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Guikun Chen, Jin Li, Wenguan Wang","submitted_at":"2024-10-20T11:40:31Z","abstract_excerpt":"Current approaches for open-vocabulary scene graph generation (OVSGG) use vision-language models such as CLIP and follow a standard zero-shot pipeline -- computing similarity between the query image and the text embeddings for each category (i.e., text classifiers). In this work, we argue that the text classifiers adopted by existing OVSGG methods, i.e., category-/part-level prompts, are scene-agnostic as they remain unchanged across contexts. Using such fixed text classifiers not only struggles to model visual relations with high variance, but also falls short in adapting to distinct contexts"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15364","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15364/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15364","created_at":"2026-07-05T09:23:18.403742+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15364v1","created_at":"2026-07-05T09:23:18.403742+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15364","created_at":"2026-07-05T09:23:18.403742+00:00"},{"alias_kind":"pith_short_12","alias_value":"HUPC5NNAPCMN","created_at":"2026-07-05T09:23:18.403742+00:00"},{"alias_kind":"pith_short_16","alias_value":"HUPC5NNAPCMNBQMI","created_at":"2026-07-05T09:23:18.403742+00:00"},{"alias_kind":"pith_short_8","alias_value":"HUPC5NNA","created_at":"2026-07-05T09:23:18.403742+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.21788","citing_title":"SceneGraphGrounder: Zero-Shot 3D Visual Grounding via Structured Scene Graph Matching","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH","json":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH.json","graph_json":"https://pith.science/api/pith-number/HUPC5NNAPCMNBQMIGWNGJIE4WH/graph.json","events_json":"https://pith.science/api/pith-number/HUPC5NNAPCMNBQMIGWNGJIE4WH/events.json","paper":"https://pith.science/paper/HUPC5NNA"},"agent_actions":{"view_html":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH","download_json":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH.json","view_paper":"https://pith.science/paper/HUPC5NNA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15364&json=true","fetch_graph":"https://pith.science/api/pith-number/HUPC5NNAPCMNBQMIGWNGJIE4WH/graph.json","fetch_events":"https://pith.science/api/pith-number/HUPC5NNAPCMNBQMIGWNGJIE4WH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH/action/storage_attestation","attest_author":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH/action/author_attestation","sign_citation":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH/action/citation_signature","submit_replication":"https://pith.science/pith/HUPC5NNAPCMNBQMIGWNGJIE4WH/action/replication_record"}},"created_at":"2026-07-05T09:23:18.403742+00:00","updated_at":"2026-07-05T09:23:18.403742+00:00"}