{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:64MJQ7256H5PSQLQ6YMF7MG66R","short_pith_number":"pith:64MJQ725","schema_version":"1.0","canonical_sha256":"f718987f5df1faf94170f6185fb0def4561bb4b38ad4d4a80ab4b1aee3fb861a","source":{"kind":"arxiv","id":"2501.09041","version":1},"attestation_state":"computed","paper":{"title":"Generative Visual Commonsense Answering and Explaining with Generative Scene Graph Constructing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Fan Yuan, Jing Li, Piji Li, Rong Quan, Wei Bi, Xiaogang Xu, Xiaoyuan Fang","submitted_at":"2025-01-15T04:00:36Z","abstract_excerpt":"Visual Commonsense Reasoning, which is regarded as one challenging task to pursue advanced visual scene comprehension, has been used to diagnose the reasoning ability of AI systems. However, reliable reasoning requires a good grasp of the scene's details. Existing work fails to effectively exploit the real-world object relationship information present within the scene, and instead overly relies on knowledge from training memory. Based on these observations, we propose a novel scene-graph-enhanced visual commonsense reasoning generation method named \\textit{\\textbf{G2}}, which first utilizes th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09041","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-15T04:00:36Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"dd1bdaffa479f9da0048fe22d72bef1cd1c5161cd16df9c3299ae65f0fa2e9cf","abstract_canon_sha256":"31227407c7128d8bc82aa8578be71a1b8958c8a9dedaed1f4af5c5a8cbf2745b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:01:40.828373Z","signature_b64":"cb8eTK4A4LNRQICaxAytn1beG9Y1v83+FMpNMx+Et5aqy6s1+b6X7vZ6SfKCn+lO/NTZNilEDhAXNTgduNDsBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f718987f5df1faf94170f6185fb0def4561bb4b38ad4d4a80ab4b1aee3fb861a","last_reissued_at":"2026-07-05T10:01:40.827923Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:01:40.827923Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generative Visual Commonsense Answering and Explaining with Generative Scene Graph Constructing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Fan Yuan, Jing Li, Piji Li, Rong Quan, Wei Bi, Xiaogang Xu, Xiaoyuan Fang","submitted_at":"2025-01-15T04:00:36Z","abstract_excerpt":"Visual Commonsense Reasoning, which is regarded as one challenging task to pursue advanced visual scene comprehension, has been used to diagnose the reasoning ability of AI systems. However, reliable reasoning requires a good grasp of the scene's details. Existing work fails to effectively exploit the real-world object relationship information present within the scene, and instead overly relies on knowledge from training memory. Based on these observations, we propose a novel scene-graph-enhanced visual commonsense reasoning generation method named \\textit{\\textbf{G2}}, which first utilizes th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09041","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09041/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09041","created_at":"2026-07-05T10:01:40.827981+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09041v1","created_at":"2026-07-05T10:01:40.827981+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09041","created_at":"2026-07-05T10:01:40.827981+00:00"},{"alias_kind":"pith_short_12","alias_value":"64MJQ7256H5P","created_at":"2026-07-05T10:01:40.827981+00:00"},{"alias_kind":"pith_short_16","alias_value":"64MJQ7256H5PSQLQ","created_at":"2026-07-05T10:01:40.827981+00:00"},{"alias_kind":"pith_short_8","alias_value":"64MJQ725","created_at":"2026-07-05T10:01:40.827981+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.09695","citing_title":"Assessing Privacy Preservation and Utility in Online Vision-Language Models","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R","json":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R.json","graph_json":"https://pith.science/api/pith-number/64MJQ7256H5PSQLQ6YMF7MG66R/graph.json","events_json":"https://pith.science/api/pith-number/64MJQ7256H5PSQLQ6YMF7MG66R/events.json","paper":"https://pith.science/paper/64MJQ725"},"agent_actions":{"view_html":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R","download_json":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R.json","view_paper":"https://pith.science/paper/64MJQ725","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09041&json=true","fetch_graph":"https://pith.science/api/pith-number/64MJQ7256H5PSQLQ6YMF7MG66R/graph.json","fetch_events":"https://pith.science/api/pith-number/64MJQ7256H5PSQLQ6YMF7MG66R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R/action/storage_attestation","attest_author":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R/action/author_attestation","sign_citation":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R/action/citation_signature","submit_replication":"https://pith.science/pith/64MJQ7256H5PSQLQ6YMF7MG66R/action/replication_record"}},"created_at":"2026-07-05T10:01:40.827981+00:00","updated_at":"2026-07-05T10:01:40.827981+00:00"}