{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CGPD5XRMJ7E4BOWUS2CWBMNF65","short_pith_number":"pith:CGPD5XRM","schema_version":"1.0","canonical_sha256":"119e3ede2c4fc9c0bad4968560b1a5f7604f760af8fe3159bce5cfe65a6c0c06","source":{"kind":"arxiv","id":"2305.17497","version":2},"attestation_state":"computed","paper":{"title":"FACTUAL: A Benchmark for Faithful and Consistent Textual Scene Graph Parsing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Donghong Ji, Fei Li, Gholamreza Haffari, Lizhen Qu, Quan Hung Tran, Terry Yue Zhuo, Yuyang Chai, Zhuang Li","submitted_at":"2023-05-27T15:38:31Z","abstract_excerpt":"Textual scene graph parsing has become increasingly important in various vision-language applications, including image caption evaluation and image retrieval. However, existing scene graph parsers that convert image captions into scene graphs often suffer from two types of errors. First, the generated scene graphs fail to capture the true semantics of the captions or the corresponding images, resulting in a lack of faithfulness. Second, the generated scene graphs have high inconsistency, with the same semantics represented by different annotations.\n  To address these challenges, we propose a n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.17497","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-27T15:38:31Z","cross_cats_sorted":[],"title_canon_sha256":"ee97fe37c360a67bc2cccf7b6a74bfa081435d835a9cef06d39313dd13594db8","abstract_canon_sha256":"f16458953fba976c80668164c059f762b4ce662f68a7812148a348ae51ab4bda"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:19.078824Z","signature_b64":"q8deRpQ5ZV0qmZZrphXQLxJ8oWFM1yvdBfVqjWz1X1IGcli94QqF4Kmt+WA/mbywtCBV8QT/u3Do6lNNSDkTAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"119e3ede2c4fc9c0bad4968560b1a5f7604f760af8fe3159bce5cfe65a6c0c06","last_reissued_at":"2026-07-05T06:16:19.078331Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:19.078331Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FACTUAL: A Benchmark for Faithful and Consistent Textual Scene Graph Parsing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Donghong Ji, Fei Li, Gholamreza Haffari, Lizhen Qu, Quan Hung Tran, Terry Yue Zhuo, Yuyang Chai, Zhuang Li","submitted_at":"2023-05-27T15:38:31Z","abstract_excerpt":"Textual scene graph parsing has become increasingly important in various vision-language applications, including image caption evaluation and image retrieval. However, existing scene graph parsers that convert image captions into scene graphs often suffer from two types of errors. First, the generated scene graphs fail to capture the true semantics of the captions or the corresponding images, resulting in a lack of faithfulness. Second, the generated scene graphs have high inconsistency, with the same semantics represented by different annotations.\n  To address these challenges, we propose a n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.17497","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.17497/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.17497","created_at":"2026-07-05T06:16:19.078388+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.17497v2","created_at":"2026-07-05T06:16:19.078388+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.17497","created_at":"2026-07-05T06:16:19.078388+00:00"},{"alias_kind":"pith_short_12","alias_value":"CGPD5XRMJ7E4","created_at":"2026-07-05T06:16:19.078388+00:00"},{"alias_kind":"pith_short_16","alias_value":"CGPD5XRMJ7E4BOWU","created_at":"2026-07-05T06:16:19.078388+00:00"},{"alias_kind":"pith_short_8","alias_value":"CGPD5XRM","created_at":"2026-07-05T06:16:19.078388+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.12455","citing_title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17118","citing_title":"Differentiable Optimization Layers for Guaranteed Fairness in Deep Learning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16538","citing_title":"VC-Inspector: Advancing Reference-free Evaluation of Video Captions with Factual Analysis","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14069","citing_title":"Towards Unconstrained Human-Object Interaction","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65","json":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65.json","graph_json":"https://pith.science/api/pith-number/CGPD5XRMJ7E4BOWUS2CWBMNF65/graph.json","events_json":"https://pith.science/api/pith-number/CGPD5XRMJ7E4BOWUS2CWBMNF65/events.json","paper":"https://pith.science/paper/CGPD5XRM"},"agent_actions":{"view_html":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65","download_json":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65.json","view_paper":"https://pith.science/paper/CGPD5XRM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.17497&json=true","fetch_graph":"https://pith.science/api/pith-number/CGPD5XRMJ7E4BOWUS2CWBMNF65/graph.json","fetch_events":"https://pith.science/api/pith-number/CGPD5XRMJ7E4BOWUS2CWBMNF65/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65/action/storage_attestation","attest_author":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65/action/author_attestation","sign_citation":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65/action/citation_signature","submit_replication":"https://pith.science/pith/CGPD5XRMJ7E4BOWUS2CWBMNF65/action/replication_record"}},"created_at":"2026-07-05T06:16:19.078388+00:00","updated_at":"2026-07-05T06:16:19.078388+00:00"}