{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:TIKUHHW4XE3532OMFUL3ZS6H2M","short_pith_number":"pith:TIKUHHW4","schema_version":"1.0","canonical_sha256":"9a15439edcb937dde9cc2d17bccbc7d32d7b233f55c0e519d2809a9094263202","source":{"kind":"arxiv","id":"1704.08883","version":2},"attestation_state":"computed","paper":{"title":"Traffic Light Control Using Deep Policy-Gradient and Value-Function Based Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Enda Howley, Michael Schukat, Seyed Sajad Mousavi","submitted_at":"2017-04-28T11:44:42Z","abstract_excerpt":"Recent advances in combining deep neural network architectures with reinforcement learning techniques have shown promising potential results in solving complex control problems with high dimensional state and action spaces. Inspired by these successes, in this paper, we build two kinds of reinforcement learning algorithms: deep policy-gradient and value-function based agents which can predict the best possible traffic signal for a traffic intersection. At each time step, these adaptive traffic light control agents receive a snapshot of the current state of a graphical traffic simulator and pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1704.08883","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-04-28T11:44:42Z","cross_cats_sorted":[],"title_canon_sha256":"98625a526568dcdfb492222aef76ce3e843fa67aabe6039a4de5b49d9aea7e4a","abstract_canon_sha256":"721da4eac32a65d4c6e0738a4f37f063e4085dbccf8804be263ef118d4a4d445"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:43:34.978279Z","signature_b64":"5/tWIU6gFMwk0WAWAhK+a42WtxHR93Yg0WV5UMNhzaLfSGOeW1oXiZrI3Zt8RA1B6y3lyHkD6g3xv95ZmYzfBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a15439edcb937dde9cc2d17bccbc7d32d7b233f55c0e519d2809a9094263202","last_reissued_at":"2026-05-18T00:43:34.977874Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:43:34.977874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Traffic Light Control Using Deep Policy-Gradient and Value-Function Based Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Enda Howley, Michael Schukat, Seyed Sajad Mousavi","submitted_at":"2017-04-28T11:44:42Z","abstract_excerpt":"Recent advances in combining deep neural network architectures with reinforcement learning techniques have shown promising potential results in solving complex control problems with high dimensional state and action spaces. Inspired by these successes, in this paper, we build two kinds of reinforcement learning algorithms: deep policy-gradient and value-function based agents which can predict the best possible traffic signal for a traffic intersection. At each time step, these adaptive traffic light control agents receive a snapshot of the current state of a graphical traffic simulator and pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1704.08883","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1704.08883","created_at":"2026-05-18T00:43:34.977932+00:00"},{"alias_kind":"arxiv_version","alias_value":"1704.08883v2","created_at":"2026-05-18T00:43:34.977932+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1704.08883","created_at":"2026-05-18T00:43:34.977932+00:00"},{"alias_kind":"pith_short_12","alias_value":"TIKUHHW4XE35","created_at":"2026-05-18T12:31:46.661854+00:00"},{"alias_kind":"pith_short_16","alias_value":"TIKUHHW4XE3532OM","created_at":"2026-05-18T12:31:46.661854+00:00"},{"alias_kind":"pith_short_8","alias_value":"TIKUHHW4","created_at":"2026-05-18T12:31:46.661854+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"1909.00395","citing_title":"An Open-Source Framework for Adaptive Traffic Signal Control","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M","json":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M.json","graph_json":"https://pith.science/api/pith-number/TIKUHHW4XE3532OMFUL3ZS6H2M/graph.json","events_json":"https://pith.science/api/pith-number/TIKUHHW4XE3532OMFUL3ZS6H2M/events.json","paper":"https://pith.science/paper/TIKUHHW4"},"agent_actions":{"view_html":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M","download_json":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M.json","view_paper":"https://pith.science/paper/TIKUHHW4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1704.08883&json=true","fetch_graph":"https://pith.science/api/pith-number/TIKUHHW4XE3532OMFUL3ZS6H2M/graph.json","fetch_events":"https://pith.science/api/pith-number/TIKUHHW4XE3532OMFUL3ZS6H2M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M/action/storage_attestation","attest_author":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M/action/author_attestation","sign_citation":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M/action/citation_signature","submit_replication":"https://pith.science/pith/TIKUHHW4XE3532OMFUL3ZS6H2M/action/replication_record"}},"created_at":"2026-05-18T00:43:34.977932+00:00","updated_at":"2026-05-18T00:43:34.977932+00:00"}