{"as_of":"2026-08-08T17:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d2a5abfc29b0e92d0f78b22f907ba408ec1f11b3cca178b3660ae0c0a2cf5865","coverage":[{"denominator":46,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":46,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:55:26.474627Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.20645/citation-record","integrity":"/paper/2505.20645/integrity","json":"/paper/2505.20645/citation-record.json","paper":"/paper/2505.20645"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:23.031214Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.031214Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:5e261ddfbcb3ea4397825ea66b346b9f09518006439f4f9c4936dc333ad3b0d0","observation_id":"3678a104-4fb6-4ba7-b583-62d3c55e9982","resolution":{"observed_at":"2026-08-07T13:55:23.031214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:23.164159Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.164159Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:12ad6d8d22ba65f186fc1b2b268e908ddee7354dc21039102fa2d3e9cac4f7f0","observation_id":"9a2f5bd2-6d8e-49ac-a2a9-73bc4c4f41c6","resolution":{"observed_at":"2026-08-07T13:55:23.164159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:23.301304Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.301304Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:fca8b1a2fadd72c7a2c50372f250e9a0dde16157fd2fd05fa8cd8f04d5ba96ec","observation_id":"3d64b0ce-6a56-4875-93c3-e0152a72c205","resolution":{"observed_at":"2026-08-07T13:55:23.301304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.00177","last_updated":"2025-02-28T20:43:45Z","snapshot_observed_at":"2026-08-07T17:38:20.897232Z","submitted_at":"2025-02-28T20:43:45Z","title":"Steering Large Language Model Activations in Sparse Spaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.00177","snapshot_observed_at":"2026-08-07T13:55:23.396550Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.396550Z"},"links":{"cited_paper":"/paper/2503.00177","citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:40f5189dc023b5f6b7c95b8c072c28900f4887c9bf5a40a2ea31018d05288625","observation_id":"7fed5dda-db73-4f3a-b954-95000fee73a0","resolution":{"observed_at":"2026-08-07T13:55:23.396550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:28.924500Z","title":null,"venue":null,"work_id":"c8745b62-ccb7-4f73-af02-d2cae11873fb","year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.469895Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:1927856ff5d3cc2e599ce79d507a31abc422076dfedbec5191a974fc9dd82252","observation_id":"d1e4f3fd-8550-4f8f-8ff7-57771923261d","resolution":{"observed_at":"2026-08-07T13:55:28.967099Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:28.685748Z","title":null,"venue":null,"work_id":"18a5f478-9e01-44c8-b6d5-33edfa7ffb89","year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.544602Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:0df598a95f49aeba8134a343a9c6de2440360406ad39de74f7a67f89e774dc8d","observation_id":"f4a29e80-2618-4887-a8af-bbaff5786fda","resolution":{"observed_at":"2026-08-07T13:55:28.779620Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:23.646609Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.646609Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:2d9fb7cd262eaa31cefa51f1891362f47c59a2901a5eeb204453ae90020ecece","observation_id":"03509357-a1a6-4528-88c9-eafe6a1bcaf0","resolution":{"observed_at":"2026-08-07T13:55:23.646609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11484","last_updated":"2025-01-10T02:18:01Z","snapshot_observed_at":"2026-08-07T19:18:05.527101Z","submitted_at":"2024-07-16T08:20:39Z","title":"The Oscars of AI Theater: A Survey on Role-Playing with Language Models","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11484","snapshot_observed_at":"2026-08-07T13:55:23.756147Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.756147Z"},"links":{"cited_paper":"/paper/2407.11484","citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:e62edc2bb56e79fb4b52d7852826b84dbddee2fc1eb66a6631c010e7557bea37","observation_id":"6b99d4f3-d131-4f29-8441-94681a182eee","resolution":{"observed_at":"2026-08-07T13:55:23.756147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:23.877640Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.877640Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:5bda1893c1312727621b1ea7858c6399f055aab269a7f27b7d21917fd400ff86","observation_id":"f0844eaf-74da-4531-84f5-26950888537f","resolution":{"observed_at":"2026-08-07T13:55:23.877640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:28.509905Z","title":null,"venue":null,"work_id":"dd6e2295-2497-4359-9863-060224edc219","year":2019},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:23.959641Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:91f31a7c2bb18f079f3a16ac85db7352a18de8c749f4d1964b6970521a735f28","observation_id":"9189b25b-60b9-4cc0-b65c-cbecb2670ef6","resolution":{"observed_at":"2026-08-07T13:55:28.571705Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:24.021433Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.021433Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:d3ea4960a18805e4467f14fcf493d734222807238722f37d592ee8e596a6affc","observation_id":"08ccbc95-eb83-45ba-835a-c09817365ff4","resolution":{"observed_at":"2026-08-07T13:55:24.021433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:28.320138Z","title":null,"venue":null,"work_id":"cb069434-5871-4c3b-a6dc-f49dfc0e399d","year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.105466Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:7a739f679384102da45f2bf1741f6dc85c8b24da8820d2cba9730bd35f2cea22","observation_id":"f29105e1-6cb2-4ccb-ad53-eef96b8fe982","resolution":{"observed_at":"2026-08-07T13:55:28.402652Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.05794","last_updated":"2022-03-11T08:35:15Z","snapshot_observed_at":"2026-07-06T12:46:41.057346Z","submitted_at":"2022-03-11T08:35:15Z","title":"BERTopic: Neural topic modeling with a class-based TF-IDF procedure","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.05794","snapshot_observed_at":"2026-08-07T13:55:24.177954Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.177954Z"},"links":{"cited_paper":"/paper/2203.05794","citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:cb37c98ddcf62977ebee85b9936796ee8cb8831098b17efc79ce40af9855aeab","observation_id":"806d745c-ddd5-408f-82f9-1597793be352","resolution":{"observed_at":"2026-08-07T13:55:24.177954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:24.215136Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.215136Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:95b5f3924ae1e51df17ffe7de7421c73d697519ab5571b7f54a2b058fc2697a7","observation_id":"2bbf2f84-1d77-4032-956e-52218ecdd21e","resolution":{"observed_at":"2026-08-07T13:55:24.215136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:28.228724Z","title":"a m \\\"a l \\","venue":null,"work_id":"20331a72-44ba-45e5-b283-a89d3064985c","year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.265170Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:972fba58e2e1e7eabaebd4168b5ad777cc9f3ff25458202f63e283f019e33504","observation_id":"bd7f1696-27dc-478c-a417-f14fa3e506e5","resolution":{"observed_at":"2026-08-07T13:55:28.260944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15846","last_updated":"2024-06-18T13:16:36Z","snapshot_observed_at":"2026-08-03T19:57:23.894721Z","submitted_at":"2024-04-24T12:51:14Z","title":"From Complex to Simple: Enhancing Multi-Constraint Complex Instruction Following Ability of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15846","snapshot_observed_at":"2026-08-07T13:55:24.305287Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.305287Z"},"links":{"cited_paper":"/paper/2404.15846","citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:653b842e1918d0c1254e946baf570072a648a243cf7283d342d3cfed1142730d","observation_id":"c524bf74-e85a-49cd-9876-4a8308cdcc8a","resolution":{"observed_at":"2026-08-07T13:55:24.305287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:24.371150Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.371150Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:57faa46808be5379ba50e193a312249e6c631c426b954ac497f59d69595832b3","observation_id":"d5c68d77-fab0-40a9-93df-378883fa50d9","resolution":{"observed_at":"2026-08-07T13:55:24.371150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.findings-acl.395","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:26.620587Z","title":null,"venue":null,"work_id":"a4e76a1f-be31-420d-8254-6150c2a45746","year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.449316Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:bf9f306287ffa4d6c4c78cb7fd15332ea9f534f2175992c3f299151228689927","observation_id":"6829a3cd-4bac-4e58-a786-34465323e727","resolution":{"observed_at":"2026-08-07T13:55:26.696008Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:28.085703Z","title":"Aligning ai with shared human values","venue":null,"work_id":"f31f6056-f794-42fa-a1d6-e3cbdb3bf595","year":null},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.514927Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:bad5b40e7ccaf7d3e32aab3ca63a8483f0ff5aee1fe7690e48504652c0a25afd","observation_id":"341012a3-0db8-46d9-b2bb-bd601dab9778","resolution":{"observed_at":"2026-08-07T13:55:28.164053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:27.985088Z","title":"Smith, and Hannaneh Hajishirzi","venue":null,"work_id":"628485ca-95da-42d0-8a25-e5c8e1598630","year":2025},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.591601Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:55b5643a01f47e5a07cccd39fa1e6addc8425efe3fa80a9414ae6109317d8519","observation_id":"0201ebbc-74d1-4525-8e8c-ce6b79439072","resolution":{"observed_at":"2026-08-07T13:55:28.021854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:24.666573Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.666573Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:736b66dc750d6ed9356f4a47282778d45f092ebb0a7ee58b9d05a48be9adaad3","observation_id":"142e94a9-8af7-4201-9cf1-f593f5170a7f","resolution":{"observed_at":"2026-08-07T13:55:24.666573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:24.764658Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.764658Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:d6631d16319bddf03442a7d6c14570fca365d1dd9eb92f664647de38aa6c30ca","observation_id":"320203d5-3a3e-43a3-bdeb-1b2f68262e7b","resolution":{"observed_at":"2026-08-07T13:55:24.764658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:27.859745Z","title":null,"venue":null,"work_id":"0f6fb302-112f-45f3-98af-44e8ea4ed98e","year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.838996Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:e77d486671de212d6ebc4169e3b87f392e9d0566b025055949d409bb62de685d","observation_id":"5a2905ac-aa31-4006-9fef-2a72a38038ad","resolution":{"observed_at":"2026-08-07T13:55:27.914835Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:24.895883Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.895883Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:c540cfbc794d171da9840d75012832e6097c8b27a11faad45f3540fbc2d282cc","observation_id":"5172d1ae-6a34-446f-acd1-f875e135d0b8","resolution":{"observed_at":"2026-08-07T13:55:24.895883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:27.747492Z","title":null,"venue":null,"work_id":"8273269e-869e-44a2-86f8-57c605f10ddd","year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:24.981143Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:ea23ab62aad59f5611f8bdb37fb16b1a2440711be2d144e1ec2d537bb9252aa0","observation_id":"f94c4197-7cfe-4008-8aa4-4fcea416ffd9","resolution":{"observed_at":"2026-08-07T13:55:27.796003Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.063989Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.063989Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:2e1349b2cfc1f7294d652528fc87401114dd8b4c0f732d03875f47d9c647a799","observation_id":"10feda8b-dfb9-4b64-8be4-60d2c58ca271","resolution":{"observed_at":"2026-08-07T13:55:25.063989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.167622Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.167622Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:5b85955e2332bc8c83ac2ddfa0093a8edabc226ab20bc4c38e2493ad5b28c785","observation_id":"e153c7aa-c5c8-464e-9078-e319e3e12d75","resolution":{"observed_at":"2026-08-07T13:55:25.167622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.210926Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.210926Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:fd82ec5f509bf494c9136b1540bc7653c121812152520a9badd792053bce6e2e","observation_id":"ded4939f-32f4-4a99-b7e4-6b20b583700e","resolution":{"observed_at":"2026-08-07T13:55:25.210926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.275767Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.275767Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:a9b3e78363066bda56310c2c6f6353276c57fb5922adf08c99394c3326d85c7b","observation_id":"3d982a9b-72ab-4994-9114-23c61d3905ee","resolution":{"observed_at":"2026-08-07T13:55:25.275767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.347527Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.347527Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:551d757fc93334737fd11abf388a4a0e200988b95bd61399230a0e6234bc99fe","observation_id":"bbd91e97-fe96-4c32-8a49-1cc55e6f992f","resolution":{"observed_at":"2026-08-07T13:55:25.347527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.415397Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.415397Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:578ce1a267ca4c9b0de0b4a3a7da0ec82914d6ea56b0efa19108505c92da2814","observation_id":"bce8bca1-0acd-4c7e-973a-5fff009596c1","resolution":{"observed_at":"2026-08-07T13:55:25.415397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.490965Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.490965Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:8dcfb7ecaae0089df1bb1dfc3a9467d565bb16987ab7502d63ea1dbf5a50e9b5","observation_id":"77aa0b0c-fd8b-4f74-b852-c23db2761c7c","resolution":{"observed_at":"2026-08-07T13:55:25.490965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.537083Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.537083Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:85670ace29cf00dfd66519afdff33b9d669f7212a332589aff94d13106ad1a20","observation_id":"353ef87f-509a-4f8b-864f-aa2b54501e17","resolution":{"observed_at":"2026-08-07T13:55:25.537083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.607345Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.607345Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:754a43803524913e325fd82bebabbe6604f33ed1ebf3fdd429ff8264e6553cc0","observation_id":"6f5adc31-3497-48aa-9acc-c271fc58ad3c","resolution":{"observed_at":"2026-08-07T13:55:25.607345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02823","last_updated":"2024-04-03T15:55:39Z","snapshot_observed_at":"2026-07-06T17:55:12.784422Z","submitted_at":"2024-04-03T15:55:39Z","title":"Conifer: Improving Complex Constrained Instruction-Following Ability of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02823","snapshot_observed_at":"2026-08-07T13:55:25.697100Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.697100Z"},"links":{"cited_paper":"/paper/2404.02823","citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:8b6656e88276e66fa8616aab1d6f64132c1da95b68e9fb48738b4dab529c08eb","observation_id":"e52aa194-98a4-437f-a161-ffcdd34fb001","resolution":{"observed_at":"2026-08-07T13:55:25.697100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.774971Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.774971Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:56f01f27737578fb83f8eb22073ce135f8f0b69adea04242d4f46a709907858b","observation_id":"060c3749-cb58-417e-ab6d-95e70ad157ec","resolution":{"observed_at":"2026-08-07T13:55:25.774971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.821115Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.821115Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:cb1bb6c3ff67e225cb0e359810f2e8350a948bf9a67b0c5a714db562532b552c","observation_id":"a2682378-3c5c-4aae-b735-c460e0eb1941","resolution":{"observed_at":"2026-08-07T13:55:25.821115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.853785Z","title":"Smith, Daniel Khashabi, and Hannaneh Hajishirzi","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.853785Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:456f4cb2193077334c232197287e588a4feb9b2bb0c3d49b4d3734e40ef4b84a","observation_id":"3b542f9a-63ee-458e-804a-5d4cb0dda103","resolution":{"observed_at":"2026-08-07T13:55:25.853785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:25.908255Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:25.908255Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:cd7e8b842afb52337bf2591e727c569b74541e8a20df905b0e7f27f1ee6509bd","observation_id":"7f2e6fcc-87c9-4235-8cf4-4b186d2b7140","resolution":{"observed_at":"2026-08-07T13:55:25.908255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:27.604220Z","title":null,"venue":null,"work_id":"49828a64-8633-4a12-8653-26d951b6e43e","year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:26.012147Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:cc8d56e4b825965c09287c3a275c37d63fcffe0bb2f54360abfdbdeea44cd051","observation_id":"9c1a650e-6a35-4559-a5cd-2d28fe02dd87","resolution":{"observed_at":"2026-08-07T13:55:27.657272Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:26.074596Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:26.074596Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:3ae7cd602d30ef97d5fa475c69be4aa48e3b02f90ab8ce6bc0784e16df3fccbe","observation_id":"cd076df4-4dea-48fc-95ff-a02506e5f376","resolution":{"observed_at":"2026-08-07T13:55:26.074596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:26.156757Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:26.156757Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:6ea764483ef52590d52f728a783b922edd7951b88c0c78a3c34a3e3ad1738ed3","observation_id":"96d78548-34e7-49c8-b13e-c72d4c514d40","resolution":{"observed_at":"2026-08-07T13:55:26.156757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:26.240689Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:26.240689Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:191421cadbda8666fc18d2150677455033ac2e1ed4c8a6d649f0e99d4d003f81","observation_id":"b4dfd2be-3187-4604-9d96-bade2c08ff9d","resolution":{"observed_at":"2026-08-07T13:55:26.240689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:27.467205Z","title":null,"venue":null,"work_id":"0a69c5df-3fe4-4b08-8249-350d1f5e2bbd","year":2025},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:26.293417Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:04ca9ea8816d065cf4c000c7598d99608d5a004cc72012d5557adb2fc5ff6b54","observation_id":"87026bb3-9943-4c81-b678-b9266d4d4563","resolution":{"observed_at":"2026-08-07T13:55:27.529932Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:55:27.345200Z","title":null,"venue":null,"work_id":"47b684c4-375b-407a-ae12-4ee00570d287","year":2025},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:26.380658Z"},"links":{"citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:d93c69f904d33ee9e0b02da6719ff10983c6fc4760d7ed20156b492b20df17e4","observation_id":"bc408312-14ce-4608-afd9-4918149dbf08","resolution":{"observed_at":"2026-08-07T13:55:27.402408Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-07T13:55:26.474627Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T13:55:26.474627Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2505.20645"},"observation_digest":"sha256:8c1c4c15ce6d376fd3863f031577ba5fe26f3951dfe7052ccadd66ce0f297184","observation_id":"46d49f00-0b25-4345-9b97-5c13ef782fda","resolution":{"observed_at":"2026-08-07T13:55:26.474627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.20645","last_updated":"2025-06-04T06:18:10Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T13:47:38.177365Z","submitted_at":"2025-05-27T02:47:56Z","title":"STEER-BENCH: A Benchmark for Evaluating the Steerability of Large Language Models"},"reference_resolution":{"displayed":46,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":42,"verified_exact":1,"verified_fuzzy":3},"total_outbound_references":46},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 46 of 46 outbound references and 0 inbound Pith citation observations for arXiv:2505.20645."}