@inproceedings{317a7ef8ef2941358272159efea59ca9,
title = "CCPO: Conservatively Constrained Policy Optimization using state augmentation",
abstract = "How to satisfy safety constraints almost surely (or with probability one) is becoming an emerging research issue for safe reinforcement learning (RL) algorithms in safety-critical domains. For instance, self-driving cars are expected to ensure that the driving strategy they adopt will never do harm to pedestrians and themselves. However, existing safe RL algorithms suffer from either risky and unstable constraint satisfaction or slow convergence. To tackle these two issues, we propose Conservatively Constrained Policy Optimization (CCPO) using state augmentation. CCPO designs a simple yet effective penalized reward function by introducing safety states and adaptive penalty factors under Safety Augmented MDP framework. Specifically, a novel Safety Promotion Function (SPF) is proposed to make the agent being more concentrated on constraint satisfaction with faster convergence by reshaping a more conservative constrained optimization objective. Moreover, we theoretically prove the convergence of CCPO. To validate both the effectiveness and efficiency of CCPO, comprehensive experiments are conducted in both single-constraint and more challenging multi-constraint environments. The experimental results demonstrate that the safe RL algorithms augmented by CCPO satisfy the predefined safety constraints almost surely and gain almost equivalent cumulative reward with faster convergence.",
author = "Zepeng Wang and Xiaochuan Shi and Chao Ma and Libing Wu and Jia Wu",
note = "Copyright the Author(s) 2023. Version archived for private and non-commercial use with the permission of the author/s and according to publisher conditions. For further rights please contact the publisher.; 26th European Conference on Artificial Intelligence, ECAI 2023 ; Conference date: 30-09-2023 Through 04-10-2023",
year = "2023",
doi = "10.3233/FAIA230566",
language = "English",
isbn = "9781643684369",
series = "Frontiers in Artificial Intelligence and Applications",
publisher = "IOS Press",
pages = "2599--2606",
editor = "Kobi Gal and Ann Now{\'e} and Nalepa, {Grzegorz J.} and Roy Fairstein and Roxana R{\u a}dulescu",
booktitle = "ECAI 2023",
address = "Netherlands",
}