@inproceedings{374a14d2aa49481dacd1fb6c6c77e68d,
title = "Avoiding wireheading with value reinforcement learning",
abstract = "How can we design good goals for arbitrarily intelligent agents? Reinforcement learning (RL) may seem like a natural approach. Unfortunately, RL does not work well for generally intelligent agents, as RL agents are incentivised to shortcut the reward sensor for maximum reward – the so-called wireheading problem. In this paper we suggest an alternative to RL called value reinforcement learning (VRL). In VRL, agents use the reward signal to learn a utility function. The VRL setup allows us to remove the incentive to wirehead by placing a constraint on the agent{\textquoteright}s actions. The constraint is defined in terms of the agent{\textquoteright}s belief distributions, and does not require an explicit specification of which actions constitute wireheading.",
author = "Tom Everitt and Marcus Hutter",
note = "Publisher Copyright: {\textcopyright} Springer International Publishing Switzerland 2016.; 9th International Conference on Artificial General Intelligence, AGI 2016 ; Conference date: 16-07-2016 Through 19-07-2016",
year = "2016",
doi = "10.1007/978-3-319-41649-6\_2",
language = "English",
isbn = "9783319416489",
series = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)",
publisher = "Springer Verlag",
pages = "12--22",
editor = "Bas Steunebrink and Pei Wang and Ben Goertzel",
booktitle = "Artificial General Intelligence - 9th International Conference, AGI 2016, Proceedings",
address = "Germany",
}