pre-release 0.10.0

This commit is contained in:
Gal Novik committed 2018-08-13 17:11:34 +03:00
1 parent d44c329bb8
commit 19ca5c24b1
485 files changed
+33292 -16770

No files matched your search

+244
View File
@@ -0,0 +1,244 @@
<!DOCTYPE html>
<!--[if IE 8]><html class="no-js lt-ie9" lang="en" > <![endif]-->
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="shortcut icon" href="/img/favicon.ico">
<title>Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="/css/theme.css" type="text/css" />
<link rel="stylesheet" href="/css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="/css/highlight.css">
<link href="/extra.css" rel="stylesheet">
<script src="/js/jquery-2.1.1.min.js"></script>
<script src="/js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="/js/highlight.pack.js"></script>
</head>
<body class="wy-body-for-nav" role="document">
<div class="wy-grid-for-nav">
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="/" class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="/search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
</form>
</div>
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<li class="toctree-l1">
<a class="" href="/">Home</a>
</li>
<li class="toctree-l1">
<a class="" href="/usage/">Usage</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li class="">
<a class="" href="/design/features/">Features</a>
</li>
<li class="">
<a class="" href="/design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="/design/network/">Network</a>
</li>
<li class="">
<a class="" href="/design/filters/">Filters</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="/algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="/algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="/algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="/algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="/algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="/algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="/algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="/algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="/algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="/dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="/contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="/contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
&nbsp;
</nav>
<section data-toggle="wy-nav-shift" class="wy-nav-content-wrap">
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="/">Reinforcement Learning Coach</a>
</nav>
<div class="wy-nav-content">
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="/">Docs</a> &raquo;</li>
<li class="wy-breadcrumbs-aside">
</li>
</ul>
<hr/>
</div>
<div role="main">
<div class="section">
<h1 id="404-page-not-found">404</h1>
<p><strong>Page not found</strong></p>
</div>
</div>
<footer>
<hr/>
<div role="contentinfo">
<!-- Copyright etc -->
</div>
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
</section>
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
</span>
</div>
<script>var base_url = '';</script>
<script src="/js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="/search/require.js"></script>
<script src="/search/search.js"></script>
</body>
</html>
View File
Whitespace-only changes.
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Behavioral Cloning - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Behavioral Cloning - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Behavioral Cloning";
var mkdocs_page_input_path = "algorithms/imitation/bc.md";
var mkdocs_page_url = "/algorithms/imitation/bc/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Behavioral Cloning</a>
<ul>
<li class="toctree-l3"><a href="#behavioral-cloning">Behavioral Cloning</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class=" current">
<a class="current" href="./">Behavioral Cloning</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#behavioral-cloning">Behavioral Cloning</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -301,10 +252,10 @@ The training goal is to reduce the difference between the actions predicted by t
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../../../dashboard/index.html" class="btn btn-neutral float-right" title="Coach Dashboard"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../../dashboard/" class="btn btn-neutral float-right" title="Coach Dashboard">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../other/dfp/index.html" class="btn btn-neutral" title="Direct Future Prediction"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../../other/dfp/" class="btn btn-neutral" title="Direct Future Prediction"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -318,7 +269,7 @@ The training goal is to reduce the difference between the actions predicted by t
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -326,17 +277,22 @@ The training goal is to reduce the difference between the actions predicted by t
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../../other/dfp/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../../other/dfp/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../../../dashboard/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../../../dashboard/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Direct Future Prediction - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Direct Future Prediction - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Direct Future Prediction";
var mkdocs_page_input_path = "algorithms/other/dfp.md";
var mkdocs_page_url = "/algorithms/other/dfp/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Direct Future Prediction</a>
<ul>
<li class="toctree-l3"><a href="#direct-future-prediction">Direct Future Prediction</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class=" current">
<a class="current" href="./">Direct Future Prediction</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#direct-future-prediction">Direct Future Prediction</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -302,10 +253,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../../imitation/bc/index.html" class="btn btn-neutral float-right" title="Behavioral Cloning"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../imitation/bc/" class="btn btn-neutral float-right" title="Behavioral Cloning">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../policy_optimization/cppo/index.html" class="btn btn-neutral" title="Clipped Proximal Policy Optimization"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../../policy_optimization/cppo/" class="btn btn-neutral" title="Clipped Proximal Policy Optimization"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -319,7 +270,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -327,17 +278,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../../policy_optimization/cppo/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../../policy_optimization/cppo/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../../imitation/bc/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../../imitation/bc/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Actor-Critic - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Actor-Critic - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Actor-Critic";
var mkdocs_page_input_path = "algorithms/policy_optimization/ac.md";
var mkdocs_page_url = "/algorithms/policy_optimization/ac/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Actor-Critic</a>
<ul>
<li class="toctree-l3"><a href="#actor-critic">Actor-Critic</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../pg/">Policy Gradient</a>
</li>
<li class=" current">
<a class="current" href="./">Actor-Critic</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#actor-critic">Actor-Critic</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -302,10 +253,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../ddpg/index.html" class="btn btn-neutral float-right" title="Deep Determinstic Policy Gradients"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../ddpg/" class="btn btn-neutral float-right" title="Deep Determinstic Policy Gradients">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../pg/index.html" class="btn btn-neutral" title="Policy Gradient"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../pg/" class="btn btn-neutral" title="Policy Gradient"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -319,7 +270,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -327,17 +278,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../pg/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../pg/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../ddpg/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../ddpg/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Clipped Proximal Policy Optimization - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Clipped Proximal Policy Optimization - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Clipped Proximal Policy Optimization";
var mkdocs_page_input_path = "algorithms/policy_optimization/cppo.md";
var mkdocs_page_url = "/algorithms/policy_optimization/cppo/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Clipped Proximal Policy Optimization</a>
<ul>
<li class="toctree-l3"><a href="#clipped-proximal-policy-optimization">Clipped Proximal Policy Optimization</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../ppo/">Proximal Policy Optimization</a>
</li>
<li class=" current">
<a class="current" href="./">Clipped Proximal Policy Optimization</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#clipped-proximal-policy-optimization">Clipped Proximal Policy Optimization</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -312,10 +263,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../../other/dfp/index.html" class="btn btn-neutral float-right" title="Direct Future Prediction"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../other/dfp/" class="btn btn-neutral float-right" title="Direct Future Prediction">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../ppo/index.html" class="btn btn-neutral" title="Proximal Policy Optimization"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../ppo/" class="btn btn-neutral" title="Proximal Policy Optimization"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -329,7 +280,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -337,17 +288,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../ppo/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../ppo/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../../other/dfp/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../../other/dfp/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Deep Determinstic Policy Gradients - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Deep Determinstic Policy Gradients - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Deep Determinstic Policy Gradients";
var mkdocs_page_input_path = "algorithms/policy_optimization/ddpg.md";
var mkdocs_page_url = "/algorithms/policy_optimization/ddpg/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Deep Determinstic Policy Gradients</a>
<ul>
<li class="toctree-l3"><a href="#deep-deterministic-policy-gradient">Deep Deterministic Policy Gradient</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../ac/">Actor-Critic</a>
</li>
<li class=" current">
<a class="current" href="./">Deep Determinstic Policy Gradients</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#deep-deterministic-policy-gradient">Deep Deterministic Policy Gradient</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -310,10 +261,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../ppo/index.html" class="btn btn-neutral float-right" title="Proximal Policy Optimization"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../ppo/" class="btn btn-neutral float-right" title="Proximal Policy Optimization">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../ac/index.html" class="btn btn-neutral" title="Actor-Critic"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../ac/" class="btn btn-neutral" title="Actor-Critic"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -327,7 +278,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -335,17 +286,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../ac/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../ac/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../ppo/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../ppo/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Policy Gradient - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Policy Gradient - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Policy Gradient";
var mkdocs_page_input_path = "algorithms/policy_optimization/pg.md";
var mkdocs_page_url = "/algorithms/policy_optimization/pg/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Policy Gradient</a>
<ul>
<li class="toctree-l3"><a href="#policy-gradient">Policy Gradient</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class=" current">
<a class="current" href="./">Policy Gradient</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#policy-gradient">Policy Gradient</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -302,10 +253,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../ac/index.html" class="btn btn-neutral float-right" title="Actor-Critic"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../ac/" class="btn btn-neutral float-right" title="Actor-Critic">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../value_optimization/naf/index.html" class="btn btn-neutral" title="Normalized Advantage Functions"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../../value_optimization/naf/" class="btn btn-neutral" title="Normalized Advantage Functions"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -319,7 +270,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -327,17 +278,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../../value_optimization/naf/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../../value_optimization/naf/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../ac/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../ac/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Proximal Policy Optimization - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Proximal Policy Optimization - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Proximal Policy Optimization";
var mkdocs_page_input_path = "algorithms/policy_optimization/ppo.md";
var mkdocs_page_url = "/algorithms/policy_optimization/ppo/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Proximal Policy Optimization</a>
<ul>
<li class="toctree-l3"><a href="#proximal-policy-optimization">Proximal Policy Optimization</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class=" current">
<a class="current" href="./">Proximal Policy Optimization</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#proximal-policy-optimization">Proximal Policy Optimization</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -303,10 +254,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../cppo/index.html" class="btn btn-neutral float-right" title="Clipped Proximal Policy Optimization"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../cppo/" class="btn btn-neutral float-right" title="Clipped Proximal Policy Optimization">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../ddpg/index.html" class="btn btn-neutral" title="Deep Determinstic Policy Gradients"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../ddpg/" class="btn btn-neutral" title="Deep Determinstic Policy Gradients"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -320,7 +271,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -328,17 +279,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../ddpg/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../ddpg/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../cppo/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../cppo/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Bootstrapped DQN - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Bootstrapped DQN - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Bootstrapped DQN";
var mkdocs_page_input_path = "algorithms/value_optimization/bs_dqn.md";
var mkdocs_page_url = "/algorithms/value_optimization/bs_dqn/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Bootstrapped DQN</a>
<ul>
<li class="toctree-l3"><a href="#bootstrapped-dqn">Bootstrapped DQN</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class=" current">
<a class="current" href="./">Bootstrapped DQN</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#bootstrapped-dqn">Bootstrapped DQN</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -304,10 +255,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../n_step/index.html" class="btn btn-neutral float-right" title="N-Step Q Learning"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../n_step/" class="btn btn-neutral float-right" title="N-Step Q Learning">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../nec/index.html" class="btn btn-neutral" title="Neural Episodic Control"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../nec/" class="btn btn-neutral" title="Neural Episodic Control"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -321,7 +272,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -329,17 +280,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../nec/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../nec/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../n_step/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../n_step/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Categorical DQN - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Categorical DQN - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Categorical DQN";
var mkdocs_page_input_path = "algorithms/value_optimization/categorical_dqn.md";
var mkdocs_page_url = "/algorithms/value_optimization/categorical_dqn/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Categorical DQN</a>
<ul>
<li class="toctree-l3"><a href="#categorical-dqn">Categorical DQN</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class=" current">
<a class="current" href="./">Categorical DQN</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#categorical-dqn">Categorical DQN</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -313,10 +264,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../mmc/index.html" class="btn btn-neutral float-right" title="Mixed Monte Carlo"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../mmc/" class="btn btn-neutral float-right" title="Mixed Monte Carlo">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../dueling_dqn/index.html" class="btn btn-neutral" title="Dueling DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../dueling_dqn/" class="btn btn-neutral" title="Dueling DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -330,7 +281,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -338,17 +289,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../dueling_dqn/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../dueling_dqn/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../mmc/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../mmc/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Double DQN - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Double DQN - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Double DQN";
var mkdocs_page_input_path = "algorithms/value_optimization/double_dqn.md";
var mkdocs_page_url = "/algorithms/value_optimization/double_dqn/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Double DQN</a>
<ul>
<li class="toctree-l3"><a href="#double-dqn">Double DQN</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class=" current">
<a class="current" href="./">Double DQN</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#double-dqn">Double DQN</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -308,10 +259,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../dueling_dqn/index.html" class="btn btn-neutral float-right" title="Dueling DQN"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../dueling_dqn/" class="btn btn-neutral float-right" title="Dueling DQN">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../dqn/index.html" class="btn btn-neutral" title="DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../dqn/" class="btn btn-neutral" title="DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -325,7 +276,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -333,17 +284,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../dqn/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../dqn/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../dueling_dqn/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../dueling_dqn/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>DQN - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>DQN - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "DQN";
var mkdocs_page_input_path = "algorithms/value_optimization/dqn.md";
var mkdocs_page_url = "/algorithms/value_optimization/dqn/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">DQN</a>
<ul>
<li class="toctree-l3"><a href="#deep-q-networks">Deep Q Networks</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class=" current">
<a class="current" href="./">DQN</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#deep-q-networks">Deep Q Networks</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -307,10 +258,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../double_dqn/index.html" class="btn btn-neutral float-right" title="Double DQN"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../double_dqn/" class="btn btn-neutral float-right" title="Double DQN">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../../usage/index.html" class="btn btn-neutral" title="Usage"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../../../design/filters/" class="btn btn-neutral" title="Filters"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -324,7 +275,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -332,17 +283,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../../../usage/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../../../design/filters/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../double_dqn/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../double_dqn/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Dueling DQN - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Dueling DQN - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Dueling DQN";
var mkdocs_page_input_path = "algorithms/value_optimization/dueling_dqn.md";
var mkdocs_page_url = "/algorithms/value_optimization/dueling_dqn/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Dueling DQN</a>
<ul>
<li class="toctree-l3"><a href="#dueling-dqn">Dueling DQN</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#general-description">General Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class=" current">
<a class="current" href="./">Dueling DQN</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#dueling-dqn">Dueling DQN</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#general-description">General Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -297,10 +248,10 @@ This is especially important in environments where there are many actions to cho
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../categorical_dqn/index.html" class="btn btn-neutral float-right" title="Categorical DQN"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../categorical_dqn/" class="btn btn-neutral float-right" title="Categorical DQN">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../double_dqn/index.html" class="btn btn-neutral" title="Double DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../double_dqn/" class="btn btn-neutral" title="Double DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -314,7 +265,7 @@ This is especially important in environments where there are many actions to cho
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -322,17 +273,22 @@ This is especially important in environments where there are many actions to cho
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../double_dqn/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../double_dqn/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../categorical_dqn/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../categorical_dqn/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Mixed Monte Carlo - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Mixed Monte Carlo - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Mixed Monte Carlo";
var mkdocs_page_input_path = "algorithms/value_optimization/mmc.md";
var mkdocs_page_url = "/algorithms/value_optimization/mmc/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Mixed Monte Carlo</a>
<ul>
<li class="toctree-l3"><a href="#mixed-monte-carlo">Mixed Monte Carlo</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class=" current">
<a class="current" href="./">Mixed Monte Carlo</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#mixed-monte-carlo">Mixed Monte Carlo</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -309,10 +260,10 @@ Once in every few thousand steps, copy the weights from the online network to th
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../pal/index.html" class="btn btn-neutral float-right" title="Persistent Advantage Learning"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../pal/" class="btn btn-neutral float-right" title="Persistent Advantage Learning">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../categorical_dqn/index.html" class="btn btn-neutral" title="Categorical DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../categorical_dqn/" class="btn btn-neutral" title="Categorical DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -326,7 +277,7 @@ Once in every few thousand steps, copy the weights from the online network to th
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -334,17 +285,22 @@ Once in every few thousand steps, copy the weights from the online network to th
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../categorical_dqn/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../categorical_dqn/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../pal/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../pal/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>N-Step Q Learning - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>N-Step Q Learning - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "N-Step Q Learning";
var mkdocs_page_input_path = "algorithms/value_optimization/n_step.md";
var mkdocs_page_url = "/algorithms/value_optimization/n_step/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">N-Step Q Learning</a>
<ul>
<li class="toctree-l3"><a href="#n-step-q-learning">N-Step Q Learning</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class=" current">
<a class="current" href="./">N-Step Q Learning</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#n-step-q-learning">N-Step Q Learning</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -308,10 +259,10 @@ where <script type="math/tex">k</script> is <script type="math/tex">T_{max} - St
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../naf/index.html" class="btn btn-neutral float-right" title="Normalized Advantage Functions"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../naf/" class="btn btn-neutral float-right" title="Normalized Advantage Functions">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../bs_dqn/index.html" class="btn btn-neutral" title="Bootstrapped DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../bs_dqn/" class="btn btn-neutral" title="Bootstrapped DQN"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -325,7 +276,7 @@ where <script type="math/tex">k</script> is <script type="math/tex">T_{max} - St
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -333,17 +284,22 @@ where <script type="math/tex">k</script> is <script type="math/tex">T_{max} - St
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../bs_dqn/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../bs_dqn/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../naf/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../naf/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Normalized Advantage Functions - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Normalized Advantage Functions - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Normalized Advantage Functions";
var mkdocs_page_input_path = "algorithms/value_optimization/naf.md";
var mkdocs_page_url = "/algorithms/value_optimization/naf/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Normalized Advantage Functions</a>
<ul>
<li class="toctree-l3"><a href="#normalized-advantage-functions">Normalized Advantage Functions</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class=" current">
<a class="current" href="./">Normalized Advantage Functions</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#normalized-advantage-functions">Normalized Advantage Functions</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -300,10 +251,10 @@ After every training step, use a soft update in order to copy the weights from t
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../../policy_optimization/pg/index.html" class="btn btn-neutral float-right" title="Policy Gradient"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../policy_optimization/pg/" class="btn btn-neutral float-right" title="Policy Gradient">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../n_step/index.html" class="btn btn-neutral" title="N-Step Q Learning"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../n_step/" class="btn btn-neutral" title="N-Step Q Learning"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -317,7 +268,7 @@ After every training step, use a soft update in order to copy the weights from t
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -325,17 +276,22 @@ After every training step, use a soft update in order to copy the weights from t
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../n_step/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../n_step/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../../policy_optimization/pg/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../../policy_optimization/pg/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Neural Episodic Control - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Neural Episodic Control - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Neural Episodic Control";
var mkdocs_page_input_path = "algorithms/value_optimization/nec.md";
var mkdocs_page_url = "/algorithms/value_optimization/nec/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Neural Episodic Control</a>
<ul>
<li class="toctree-l3"><a href="#neural-episodic-control">Neural Episodic Control</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../pal/">Persistent Advantage Learning</a>
</li>
<li class=" current">
<a class="current" href="./">Neural Episodic Control</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#neural-episodic-control">Neural Episodic Control</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -307,10 +258,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../bs_dqn/index.html" class="btn btn-neutral float-right" title="Bootstrapped DQN"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../bs_dqn/" class="btn btn-neutral float-right" title="Bootstrapped DQN">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../pal/index.html" class="btn btn-neutral" title="Persistent Advantage Learning"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../pal/" class="btn btn-neutral" title="Persistent Advantage Learning"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -324,7 +275,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -332,17 +283,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../pal/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../pal/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../bs_dqn/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../bs_dqn/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+151 -195
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Persistent Advantage Learning - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../../img/favicon.ico">
<title>Persistent Advantage Learning - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../../css/highlight.css">
<link href="../../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Persistent Advantage Learning";
var mkdocs_page_input_path = "algorithms/value_optimization/pal.md";
var mkdocs_page_url = "/algorithms/value_optimization/pal/";
</script>
<script src="../../../js/jquery-2.1.1.min.js"></script>
<script src="../../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,195 +45,150 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Persistent Advantage Learning</a>
<ul>
<li class="toctree-l3"><a href="#persistent-advantage-learning">Persistent Advantage Learning</a></li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../mmc/">Mixed Monte Carlo</a>
</li>
<li class=" current">
<a class="current" href="./">Persistent Advantage Learning</a>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_agent/index.html">Adding a New Agent</a>
<li class="toctree-l3"><a href="#persistent-advantage-learning">Persistent Advantage Learning</a></li>
<ul>
</li>
<li><a class="toctree-l4" href="#network-structure">Network Structure</a></li>
<li><a class="toctree-l4" href="#algorithm-description">Algorithm Description</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../../../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</li>
<li class="">
<a class="" href="../nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -249,7 +200,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../../..">Reinforcement Learning Coach Documentation</a>
<a href="../../..">Reinforcement Learning Coach</a>
</nav>
@@ -321,10 +272,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../nec/index.html" class="btn btn-neutral float-right" title="Neural Episodic Control"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../nec/" class="btn btn-neutral float-right" title="Neural Episodic Control">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../mmc/index.html" class="btn btn-neutral" title="Mixed Monte Carlo"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../mmc/" class="btn btn-neutral" title="Mixed Monte Carlo"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -338,7 +289,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -346,17 +297,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../mmc/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../mmc/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../nec/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../nec/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../../..';</script>
<script src="../../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../../search/require.js"></script>
<script src="../../../search/search.js"></script>
</body>
</html>
+191 -209
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Adding a New Agent - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../img/favicon.ico">
<title>Adding a New Agent - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../css/highlight.css">
<link href="../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Adding a New Agent";
var mkdocs_page_input_path = "contributing/add_agent.md";
var mkdocs_page_url = "/contributing/add_agent/";
</script>
<script src="../../js/jquery-2.1.1.min.js"></script>
<script src="../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,188 +45,139 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Adding a New Agent</a>
<ul>
</ul>
</li>
<li class="toctree-l1 ">
<a class="" href="../add_env/index.html">Adding a New Environment</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
<li>
</li>
<li class="toctree-l1">
<a class="" href="../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class=" current">
<a class="current" href="./">Adding a New Agent</a>
<ul class="subnav">
</ul>
</li>
<li class="">
<a class="" href="../add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -242,7 +189,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../..">Reinforcement Learning Coach Documentation</a>
<a href="../..">Reinforcement Learning Coach</a>
</nav>
@@ -273,42 +220,72 @@
<p>Coach's modularity makes adding an agent a simple and clean task, that involves the following steps:</p>
<ol>
<li>
<p>Implement your algorithm in a new file under the agents directory. The agent can inherit base classes such as <strong>ValueOptimizationAgent</strong> or <strong>ActorCriticAgent</strong>, or the more generic <strong>Agent</strong> base class.</p>
<p>Implement your algorithm in a new file. The agent can inherit base classes such as <strong>ValueOptimizationAgent</strong> or
<strong>ActorCriticAgent</strong>, or the more generic <strong>Agent</strong> base class.</p>
<ul>
<li>
<p><strong>ValueOptimizationAgent</strong>, <strong>PolicyOptimizationAgent</strong> and <strong>Agent</strong> are abstract classes.
learn_from_batch() should be overriden with the desired behavior for the algorithm being implemented. If deciding to inherit from <strong>Agent</strong>, also choose_action() should be overriden. </p>
<pre><code>def learn_from_batch(self, batch):
<li><strong>ValueOptimizationAgent</strong>, <strong>PolicyOptimizationAgent</strong> and <strong>Agent</strong> are abstract classes.
learn_from_batch() should be overriden with the desired behavior for the algorithm being implemented.
If deciding to inherit from <strong>Agent</strong>, also choose_action() should be overriden.<pre><code>def learn_from_batch(self, batch) -&gt; Tuple[float, List, List]:
"""
Given a batch of transitions, calculates their target values and updates the network.
:param batch: A list of transitions
:return: The loss of the training
:return: The total loss of the training, the loss per head and the unclipped gradients
"""
pass
def choose_action(self, curr_state, phase=RunPhase.TRAIN):
def choose_action(self, curr_state):
"""
choose an action to act with in the current episode being played. Different behavior might be exhibited when training
or testing.
:param curr_state: the current state to act upon.
:param phase: the current phase: training or testing.
:param curr_state: the current state to act upon.
:return: chosen action, some action value describing the action (q-value, probability, etc)
"""
pass
</code></pre>
</li>
<li>
<p>Make sure to add your new agent to <strong>agents/__init__.py</strong></p>
</li>
</ul>
</li>
<li>
<p>Implement your agent's specific network head, if needed, at the implementation for the framework of your choice. For example <strong>architectures/neon_components/heads.py</strong>. The head will inherit the generic base class Head.
A new output type should be added to configurations.py, and a mapping between the new head and output type should be defined in the get_output_head() function at <strong>architectures/neon_components/general_network.py</strong></p>
<p>Implement your agent's specific network head, if needed, at the implementation for the framework of your choice.
For example <strong>architectures/neon_components/heads.py</strong>. The head will inherit the generic base class Head.
A new output type should be added to configurations.py, and a mapping between the new head and output type should
be defined in the get_output_head() function at <strong>architectures/neon_components/general_network.py</strong></p>
</li>
<li>
<p>Define a new parameters class that inherits AgentParameters.
The parameters class defines all the hyperparameters for the agent, and is initialized with 4 main components:</p>
<ul>
<li><strong>algorithm</strong>: A class inheriting AlgorithmParameters which defines any algorithm specific parameters</li>
<li><strong>exploration</strong>: A class inheriting ExplorationParameters which defines the exploration policy parameters.
There are several common exploration policies built-in which you can use, and are defined under
the exploration sub directory. You can also define your own custom exploration policy.</li>
<li><strong>memory</strong>: A class inheriting MemoryParameters which defined the memory parameters.
There are several common memory types built-in which you can use, and are defined under the memories
sub directory. You can also define your own custom memory.</li>
<li><strong>networks</strong>: A dictionary defining all the networks that will be used by the agent. The keys of the dictionary
define the network name and will be used to access each network through the agent class.
The dictionary values are a class inheriting NetworkParameters, which define the network structure
and parameters.</li>
</ul>
<p>Additionally, set the path property to return the path to your agent class in the following format:</p>
<pre><code> &lt;path to python module&gt;:&lt;name of agent class&gt;
</code></pre>
<p>For example,</p>
<pre><code> class RainbowAgentParameters(AgentParameters):
def __init__(self):
super().__init__(algorithm=RainbowAlgorithmParameters(),
exploration=RainbowExplorationParameters(),
memory=RainbowMemoryParameters(),
networks={"main": RainbowNetworkParameters()})
@property
def path(self):
return 'rainbow.rainbow_agent:RainbowAgent'
</code></pre>
</li>
<li>
<p>(Optional) Define a preset using the new agent type with a given environment, and the hyper-parameters that should
be used for training on that environment.</p>
</li>
<li>Define a new configuration class at configurations.py, which includes the new agent name in the <strong>type</strong> field, the new output type in the <strong>output_types</strong> field, and assigning default values to hyperparameters.</li>
<li>(Optional) Define a preset using the new agent type with a given environment, and the hyperparameters that should be used for training on that environment.</li>
</ol>
</div>
@@ -317,10 +294,10 @@ def choose_action(self, curr_state, phase=RunPhase.TRAIN):
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../add_env/index.html" class="btn btn-neutral float-right" title="Adding a New Environment"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../add_env/" class="btn btn-neutral float-right" title="Adding a New Environment">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../dashboard/index.html" class="btn btn-neutral" title="Coach Dashboard"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../../dashboard/" class="btn btn-neutral" title="Coach Dashboard"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -334,7 +311,7 @@ def choose_action(self, curr_state, phase=RunPhase.TRAIN):
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -342,17 +319,22 @@ def choose_action(self, curr_state, phase=RunPhase.TRAIN):
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../../dashboard/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../../dashboard/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../add_env/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../add_env/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../..';</script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../search/require.js"></script>
<script src="../../search/search.js"></script>
</body>
</html>
+192 -227
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Adding a New Environment - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../../img/favicon.ico">
<title>Adding a New Environment - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../css/highlight.css">
<link href="../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Adding a New Environment";
var mkdocs_page_input_path = "contributing/add_env.md";
var mkdocs_page_url = "/contributing/add_env/";
</script>
<script src="../../js/jquery-2.1.1.min.js"></script>
<script src="../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,188 +45,145 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../..">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../../algorithms/imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../add_agent/index.html">Adding a New Agent</a>
</li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Adding a New Environment</a>
<ul>
</ul>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
<li>
</li>
<li class="toctree-l1">
<a class="" href="../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../add_agent/">Adding a New Agent</a>
</li>
<li class=" current">
<a class="current" href="./">Adding a New Environment</a>
<ul class="subnav">
<li class="toctree-l3"><a href="#using-the-openai-gym-api">Using the OpenAI Gym API</a></li>
<li class="toctree-l3"><a href="#using-the-coach-api">Using the Coach API</a></li>
</ul>
</li>
</ul>
</li>
</ul>
</div>
@@ -242,7 +195,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../..">Reinforcement Learning Coach Documentation</a>
<a href="../..">Reinforcement Learning Coach</a>
</nav>
@@ -269,74 +222,81 @@
<div class="section">
<p>Adding a new environment to Coach is as easy as solving CartPole. </p>
<p>There are essentially two ways to integrate new environments to Coach:</p>
<h2 id="using-the-openai-gym-api">Using the OpenAI Gym API</h2>
<p>If your environment is already using the OpenAI Gym API, you are already good to go.
When selecting the environment parameters in the preset, use GymEnvironmentParameters(),
and pass the path to your environment source code using the level parameter.
You can specify additional parameters for your environment using the additional_simulator_parameters parameter.
Take for example the definition used in the Pendulum_HAC preset:</p>
<pre><code> env_params = GymEnvironmentParameters()
env_params.level = "rl_coach.environments.mujoco.pendulum_with_goals:PendulumWithGoals"
env_params.additional_simulator_parameters = {"time_limit": 1000}
</code></pre>
<h2 id="using-the-coach-api">Using the Coach API</h2>
<p>There are a few simple steps to follow, and we will walk through them one by one.</p>
<ol>
<li>
<p>Coach defines a simple API for implementing a new environment which is defined in environment/environment_wrapper.py.
There are several functions to implement, but only some of them are mandatory. </p>
<p>Create a new class for your environment, and inherit the Environment class.</p>
</li>
<li>
<p>Coach defines a simple API for implementing a new environment, which are defined in environment/environment.py.
There are several functions to implement, but only some of them are mandatory.</p>
<p>Here are the important ones:</p>
<pre><code> def _take_action(self, action_idx):
<pre><code> def _take_action(self, action_idx: ActionType) -&gt; None:
"""
An environment dependent function that sends an action to the simulator.
:param action_idx: the action to perform on the environment.
:param action_idx: the action to perform on the environment
:return: None
"""
pass
def _preprocess_observation(self, observation):
"""
Do initial observation preprocessing such as cropping, rgb2gray, rescale etc.
Implementing this function is optional.
:param observation: a raw observation from the environment
:return: the preprocessed observation
"""
return observation
def _update_state(self):
def _update_state(self) -&gt; None:
"""
Updates the state from the environment.
Should update self.observation, self.reward, self.done, self.measurements and self.info
:return: None
"""
pass
def _restart_environment_episode(self, force_environment_reset=False):
def _restart_environment_episode(self, force_environment_reset=False) -&gt; None:
"""
Restarts the simulator episode
:param force_environment_reset: Force the environment to reset even if the episode is not done yet.
:return:
:return: None
"""
pass
def get_rendered_image(self):
def _render(self) -&gt; None:
"""
Renders the environment using the native simulator renderer
:return: None
"""
def get_rendered_image(self) -&gt; np.ndarray:
"""
Return a numpy array containing the image that will be rendered to the screen.
This can be different from the observation. For example, mujoco's observation is a measurements vector.
:return: numpy array containing the image that will be rendered to the screen
"""
return self.observation
</code></pre>
</li>
<li>
<p>Make sure to import the environment in environments/__init__.py:</p>
<pre><code>from doom_environment_wrapper import *
</code></pre>
<p>Also, a new entry should be added to the EnvTypes enum mapping the environment name to the wrapper's class name:</p>
<pre><code>Doom = "DoomEnvironmentWrapper"
<p>Create a new parameters class for your environment, which inherits the EnvironmentParameters class.
In the <strong>init</strong> of your class, define all the parameters you used in your Environment class.
Additionally, fill the path property of the class with the path to your Environment class.
For example, take a look at the EnvironmentParameters class used for Doom:</p>
<pre><code> class DoomEnvironmentParameters(EnvironmentParameters):
def __init__(self):
super().__init__()
self.default_input_filter = DoomInputFilter
self.default_output_filter = DoomOutputFilter
self.cameras = [DoomEnvironment.CameraTypes.OBSERVATION]
@property
def path(self):
return 'rl_coach.environments.doom_environment:DoomEnvironment'
</code></pre>
</li>
<li>
<p>In addition a new configuration class should be implemented for defining the environment's parameters and placed in configurations.py.
For instance, the following is used for Doom:</p>
<pre><code>class Doom(EnvironmentParameters):
type = 'Doom'
frame_skip = 4
observation_stack_size = 3
desired_observation_height = 60
desired_observation_width = 76
</code></pre>
</li>
<li>
<p>And that's it, you're done. Now just add a new preset with your newly created environment, and start training an agent on top of it. </p>
<p>And that's it, you're done. Now just add a new preset with your newly created environment, and start training an agent on top of it.</p>
</li>
</ol>
@@ -347,7 +307,7 @@ For instance, the following is used for Doom:</p>
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../add_agent/index.html" class="btn btn-neutral" title="Adding a New Agent"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../add_agent/" class="btn btn-neutral" title="Adding a New Agent"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -361,7 +321,7 @@ For instance, the following is used for Doom:</p>
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -369,15 +329,20 @@ For instance, the following is used for Doom:</p>
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../add_agent/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../add_agent/" style="color: #fcfcfc;">&laquo; Previous</a></span>
</span>
</div>
<script>var base_url = '../..';</script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../search/require.js"></script>
<script src="../../search/search.js"></script>
</body>
</html>
-1
View File
@@ -8,7 +8,6 @@ github.com style (c) Vasily Polovnyov <vast@whiteants.net>
.hljs {
display: block;
overflow-x: auto;
padding: 0.5em;
color: #333;
-webkit-text-size-adjust: none;
}
+1 -1
View File
@@ -3,7 +3,7 @@
* theme. To aid upgradability this file should *not* be edited.
* modifications we need should be included in theme_extra.css.
*
* https://github.com/rtfd/readthedocs.org/blob/master/media/css/sphinx_rtd_theme.css
* https://github.com/rtfd/readthedocs.org/blob/master/readthedocs/core/static/core/css/theme.css
*/
*{-webkit-box-sizing:border-box;-moz-box-sizing:border-box;box-sizing:border-box}article,aside,details,figcaption,figure,footer,header,hgroup,nav,section{display:block}audio,canvas,video{display:inline-block;*display:inline;*zoom:1}audio:not([controls]){display:none}[hidden]{display:none}*{-webkit-box-sizing:border-box;-moz-box-sizing:border-box;box-sizing:border-box}html{font-size:100%;-webkit-text-size-adjust:100%;-ms-text-size-adjust:100%}body{margin:0}a:hover,a:active{outline:0}abbr[title]{border-bottom:1px dotted}b,strong{font-weight:bold}blockquote{margin:0}dfn{font-style:italic}ins{background:#ff9;color:#000;text-decoration:none}mark{background:#ff0;color:#000;font-style:italic;font-weight:bold}pre,code,.rst-content tt,kbd,samp{font-family:monospace,serif;_font-family:"courier new",monospace;font-size:1em}pre{white-space:pre}q{quotes:none}q:before,q:after{content:"";content:none}small{font-size:85%}sub,sup{font-size:75%;line-height:0;position:relative;vertical-align:baseline}sup{top:-0.5em}sub{bottom:-0.25em}ul,ol,dl{margin:0;padding:0;list-style:none;list-style-image:none}li{list-style:none}dd{margin:0}img{border:0;-ms-interpolation-mode:bicubic;vertical-align:middle;max-width:100%}svg:not(:root){overflow:hidden}figure{margin:0}form{margin:0}fieldset{border:0;margin:0;padding:0}label{cursor:pointer}legend{border:0;*margin-left:-7px;padding:0;white-space:normal}button,input,select,textarea{font-size:100%;margin:0;vertical-align:baseline;*vertical-align:middle}button,input{line-height:normal}button,input[type="button"],input[type="reset"],input[type="submit"]{cursor:pointer;-webkit-appearance:button;*overflow:visible}button[disabled],input[disabled]{cursor:default}input[type="checkbox"],input[type="radio"]{box-sizing:border-box;padding:0;*width:13px;*height:13px}input[type="search"]{-webkit-appearance:textfield;-moz-box-sizing:content-box;-webkit-box-sizing:content-box;box-sizing:content-box}input[type="search"]::-webkit-search-decoration,input[type="search"]::-webkit-search-cancel-button{-webkit-appearance:none}button::-moz-focus-inner,input::-moz-focus-inner{border:0;padding:0}textarea{overflow:auto;vertical-align:top;resize:vertical}table{border-collapse:collapse;border-spacing:0}td{vertical-align:top}.chromeframe{margin:0.2em 0;background:#ccc;color:#000;padding:0.2em 0}.ir{display:block;border:0;text-indent:-999em;overflow:hidden;background-color:transparent;background-repeat:no-repeat;text-align:left;direction:ltr;*line-height:0}.ir br{display:none}.hidden{display:none !important;visibility:hidden}.visuallyhidden{border:0;clip:rect(0 0 0 0);height:1px;margin:-1px;overflow:hidden;padding:0;position:absolute;width:1px}.visuallyhidden.focusable:active,.visuallyhidden.focusable:focus{clip:auto;height:auto;margin:0;overflow:visible;position:static;width:auto}.invisible{visibility:hidden}.relative{position:relative}big,small{font-size:100%}@media print{html,body,section{background:none !important}*{box-shadow:none !important;text-shadow:none !important;filter:none !important;-ms-filter:none !important}a,a:visited{text-decoration:underline}.ir a:after,a[href^="javascript:"]:after,a[href^="#"]:after{content:""}pre,blockquote{page-break-inside:avoid}thead{display:table-header-group}tr,img{page-break-inside:avoid}img{max-width:100% !important}@page{margin:0.5cm}p,h2,h3{orphans:3;widows:3}h2,h3{page-break-after:avoid}}.fa:before,.rst-content .admonition-title:before,.rst-content h1 .headerlink:before,.rst-content h2 .headerlink:before,.rst-content h3 .headerlink:before,.rst-content h4 .headerlink:before,.rst-content h5 .headerlink:before,.rst-content h6 .headerlink:before,.rst-content dl dt .headerlink:before,.icon:before,.wy-dropdown .caret:before,.wy-inline-validate.wy-inline-validate-success .wy-input-context:before,.wy-inline-validate.wy-inline-validate-danger .wy-input-context:before,.wy-inline-validate.wy-inline-validate-warning .wy-input-context:before,.wy-inline-validate.wy-inline-validate-info .wy-input-context:before,.wy-alert,.rst-content .note,.rst-content .attention,.rst-content .caution,.rst-content .danger,.rst-content .error,.rst-content .hint,.rst-content .important,.rst-content .tip,.rst-content .warning,.rst-content .seealso,.rst-content .admonition-todo,.btn,input[type="text"],input[type="password"],input[type="email"],input[type="url"],input[type="date"],input[type="month"],input[type="time"],input[type="datetime"],input[type="datetime-local"],input[type="week"],input[type="number"],input[type="search"],input[type="tel"],input[type="color"],select,textarea,.wy-menu-vertical li.on a,.wy-menu-vertical li.current>a,.wy-side-nav-search>a,.wy-side-nav-search .wy-dropdown>a,.wy-nav-top a{-webkit-font-smoothing:antialiased}.clearfix{*zoom:1}.clearfix:before,.clearfix:after{display:table;content:""}.clearfix:after{clear:both}/*!
+97 -29
View File
@@ -1,15 +1,3 @@
/*
* Tweak the overal size to better match RTD.
*/
body {
font-size: 90%;
}
h3, h4, h5, h6 {
color: #2980b9;
font-weight: 300
}
/*
* Sphinx doesn't have support for section dividers like we do in
* MkDocs, this styles the section titles in the nav
@@ -34,10 +22,25 @@ h3, h4, h5, h6 {
* area doesn't scroll.
*
* https://github.com/mkdocs/mkdocs/pull/202
*
* Builds upon pull 202 https://github.com/mkdocs/mkdocs/pull/202
* to make toc scrollbar end before navigations buttons to not be overlapping.
*/
.wy-nav-side {
height: 100%;
height: calc(100% - 45px);
overflow-y: auto;
min-height: 0;
}
.rst-versions{
border-top: 0;
height: 45px;
}
@media screen and (max-width: 768px) {
.wy-nav-side {
height: 100%;
}
}
/*
@@ -50,23 +53,49 @@ h3, h4, h5, h6 {
margin-bottom: 2em;
}
/*
* Fix wrapping in the code highlighting
*
* https://github.com/mkdocs/mkdocs/issues/233
*/
code {
white-space: pre;
}
/*
* Wrap inline code samples otherwise they shoot of the side and
* can't be read at all.
*
* https://github.com/mkdocs/mkdocs/issues/313
* https://github.com/mkdocs/mkdocs/issues/233
* https://github.com/mkdocs/mkdocs/issues/834
*/
p code {
code {
white-space: pre-wrap;
word-wrap: break-word;
padding: 2px 5px;
}
/**
* Make code blocks display as blocks and give them the appropriate
* font size and padding.
*
* https://github.com/mkdocs/mkdocs/issues/855
* https://github.com/mkdocs/mkdocs/issues/834
* https://github.com/mkdocs/mkdocs/issues/233
*/
pre code {
white-space: pre;
word-wrap: normal;
display: block;
padding: 12px;
font-size: 12px;
}
/*
* Fix link colors when the link text is inline code.
*
* https://github.com/mkdocs/mkdocs/issues/718
*/
a code {
color: #2980B9;
}
a:hover code {
color: #3091d1;
}
a:visited code {
color: #9B59B6;
}
/*
@@ -76,7 +105,7 @@ p code {
*
* https://github.com/mkdocs/mkdocs/issues/411
*/
code.cs, code.c {
pre .cs, pre .c {
font-weight: inherit;
font-style: inherit;
}
@@ -99,21 +128,20 @@ code.cs, code.c {
* Additions specific to the search functionality provided by MkDocs
*/
#mkdocs-search-results article h3
{
.search-results article {
margin-top: 23px;
border-top: 1px solid #E1E4E5;
padding-top: 24px;
}
#mkdocs-search-results article:first-child h3 {
.search-results article:first-child {
border-top: none;
}
#mkdocs-search-query{
form .search-query {
width: 100%;
border-radius: 50px;
padding: 6px 12px;
padding: 6px 12px; /* csslint allow: box-model */
border-color: #D1D4D5;
}
@@ -124,3 +152,43 @@ code.cs, code.c {
.wy-menu-vertical li ul.subnav ul.subnav{
padding-left: 1em;
}
.wy-menu-vertical .subnav li.current > a {
padding-left: 2.42em;
}
.wy-menu-vertical .subnav li.current > ul li a {
padding-left: 3.23em;
}
/*
* Improve inline code blocks within admonitions.
*
* https://github.com/mkdocs/mkdocs/issues/656
*/
.admonition code {
color: #404040;
border: 1px solid #c7c9cb;
border: 1px solid rgba(0, 0, 0, 0.2);
background: #f8fbfd;
background: rgba(255, 255, 255, 0.7);
}
/*
* Account for wide tables which go off the side.
* Override borders to avoid wierdness on narrow tables.
*
* https://github.com/mkdocs/mkdocs/issues/834
* https://github.com/mkdocs/mkdocs/pull/1034
*/
.rst-content .section .docutils {
width: 100%;
overflow: auto;
display: block;
border: none;
}
td, th {
border: 1px solid #e1e4e5 !important; /* csslint allow: important */
border-collapse: collapse;
}
+153 -201
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Coach Dashboard - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../img/favicon.ico">
<title>Coach Dashboard - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../css/highlight.css">
<link href="../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Coach Dashboard";
var mkdocs_page_input_path = "dashboard.md";
var mkdocs_page_url = "/dashboard/";
</script>
<script src="../js/jquery-2.1.1.min.js"></script>
<script src="../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../js/highlight.pack.js"></script>
<script src="../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../index.html" class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href=".." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,197 +45,148 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../index.html">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="../usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Coach Dashboard</a>
<ul>
<li class="toctree-l3"><a href="#visualizing-signals">Visualizing Signals</a></li>
<li class="toctree-l3"><a href="#tracking-statistics">Tracking Statistics</a></li>
<li class="toctree-l3"><a href="#comparing-runs">Comparing Runs</a></li>
</ul>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../contributing/add_agent/index.html">Adding a New Agent</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../contributing/add_env/index.html">Adding a New Environment</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
<li>
</li>
<li class="toctree-l1 current">
<a class="current" href="./">Coach Dashboard</a>
<ul class="subnav">
<li class="toctree-l2"><a href="#visualizing-signals">Visualizing Signals</a></li>
<li class="toctree-l2"><a href="#tracking-statistics">Tracking Statistics</a></li>
<li class="toctree-l2"><a href="#comparing-runs">Comparing Runs</a></li>
</ul>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -251,7 +198,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../index.html">Reinforcement Learning Coach Documentation</a>
<a href="..">Reinforcement Learning Coach</a>
</nav>
@@ -259,7 +206,7 @@
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="../index.html">Docs</a> &raquo;</li>
<li><a href="..">Docs</a> &raquo;</li>
@@ -352,10 +299,10 @@
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../contributing/add_agent/index.html" class="btn btn-neutral float-right" title="Adding a New Agent"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../contributing/add_agent/" class="btn btn-neutral float-right" title="Adding a New Agent">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../algorithms/imitation/bc/index.html" class="btn btn-neutral" title="Behavioral Cloning"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href="../algorithms/imitation/bc/" class="btn btn-neutral" title="Behavioral Cloning"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -369,7 +316,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -377,17 +324,22 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../algorithms/imitation/bc/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href="../algorithms/imitation/bc/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../contributing/add_agent/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../contributing/add_agent/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '..';</script>
<script src="../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../search/require.js"></script>
<script src="../search/search.js"></script>
</body>
</html>
+367
View File
@@ -0,0 +1,367 @@
<!DOCTYPE html>
<!--[if IE 8]><html class="no-js lt-ie9" lang="en" > <![endif]-->
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="shortcut icon" href="../../img/favicon.ico">
<title>Control Flow - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../css/highlight.css">
<link href="../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Control Flow";
var mkdocs_page_input_path = "design/control_flow.md";
var mkdocs_page_url = "/design/control_flow/";
</script>
<script src="../../js/jquery-2.1.1.min.js"></script>
<script src="../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
</head>
<body class="wy-body-for-nav" role="document">
<div class="wy-grid-for-nav">
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
</form>
</div>
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<li class="toctree-l1">
<a class="" href="../..">Home</a>
</li>
<li class="toctree-l1">
<a class="" href="../../usage/">Usage</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li class="">
<a class="" href="../features/">Features</a>
</li>
<li class=" current">
<a class="current" href="./">Control Flow</a>
<ul class="subnav">
<li class="toctree-l3"><a href="#coach-control-flow">Coach Control Flow</a></li>
<ul>
<li><a class="toctree-l4" href="#graph-manager">Graph Manager</a></li>
<li><a class="toctree-l4" href="#level-manager">Level Manager</a></li>
<li><a class="toctree-l4" href="#agent">Agent</a></li>
</ul>
</ul>
</li>
<li class="">
<a class="" href="../network/">Network</a>
</li>
<li class="">
<a class="" href="../filters/">Filters</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
&nbsp;
</nav>
<section data-toggle="wy-nav-shift" class="wy-nav-content-wrap">
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../..">Reinforcement Learning Coach</a>
</nav>
<div class="wy-nav-content">
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="../..">Docs</a> &raquo;</li>
<li>Design &raquo;</li>
<li>Control Flow</li>
<li class="wy-breadcrumbs-aside">
</li>
</ul>
<hr/>
</div>
<div role="main">
<div class="section">
<!-- language-all: python -->
<h1 id="coach-control-flow">Coach Control Flow</h1>
<p>Coach is built in a modular way, encouraging modules reuse and reducing the amount of boilerplate code needed
for developing new algorithms or integrating a new challenge as an environment.
On the other hand, it can be overwhelming for new users to ramp up on the code.
To help with that, here's a short overview of the control flow.</p>
<h2 id="graph-manager">Graph Manager</h2>
<p>The main entry point for Coach is <strong>coach.py</strong>.
The main functionality of this script is to parse the command line arguments and invoke all the sub-processes needed
for the given experiment.
<strong>coach.py</strong> executes the given <strong>preset</strong> file which returns a <strong>GraphManager</strong> object.</p>
<p>A <strong>preset</strong> is a design pattern that is intended for concentrating the entire definition of an experiment in a single
file. This helps with experiments reproducibility, improves readability and prevents confusion.
The outcome of a preset is a <strong>GraphManager</strong> which will usually be instantiated in the final lines of the preset.</p>
<p>A <strong>GraphManager</strong> is an object that holds all the agents and environments of an experiment, and is mostly responsible
for scheduling their work. Why is it called a <strong>graph</strong> manager? Because agents and environments are structured into
a graph of interactions. For example, in hierarchical reinforcement learning schemes, there will often be a master
policy agent, that will control a sub-policy agent, which will interact with the environment. Other schemes can have
much more complex graphs of control, such as several hierarchy layers, each with multiple agents.
The graph manager's main loop is the improve loop.</p>
<p style="text-align: center;">
<img src="../../img/improve.png" alt="Improve loop" style="width: 400px;"/>
</p>
<p>The improve loop skips between 3 main phases - heatup, training and evaluation:</p>
<ul>
<li>
<p><strong>Heatup</strong> - the goal of this phase is to collect initial data for populating the replay buffers. The heatup phase
takes place only in the beginning of the experiment, and the agents will act completely randomly during this phase.
Importantly, the agents do not train their networks during this phase. DQN for example, uses 50k random steps in order
to initialize the replay buffers.</p>
</li>
<li>
<p><strong>Training</strong> - the training phase is the main phase of the experiment. This phase can change between agent types,
but essentially consists of repeated cycles of acting, collecting data from the environment, and training the agent
networks. During this phase, the agent will use its exploration policy in training mode, which will add noise to its
actions in order to improve its knowledge about the environment state space.</p>
</li>
<li>
<p><strong>Evaluation</strong> - the evaluation phase is intended for evaluating the current performance of the agent. The agents
will act greedily in order to exploit the knowledge aggregated so far and the performance over multiple episodes of
evaluation will be averaged in order to reduce the stochasticity effects of all the components.</p>
</li>
</ul>
<h2 id="level-manager">Level Manager</h2>
<p>In each of the 3 phases described above, the graph manager will invoke all the hierarchy levels in the graph in a
synchronized manner. In Coach, agents do not interact directly with the environment. Instead, they go through a
<em>LevelManager</em>, which is a proxy that manages their interaction. The level manager passes the current state and reward
from the environment to the agent, and the actions from the agent to the environment.</p>
<p>The motivation for having a level manager is to disentangle the code of the environment and the agent, so to allow more
complex interactions. Each level can have multiple agents which interact with the environment. Who gets to choose the
action for each step is controlled by the level manager.
Additionally, each level manager can act as an environment for the hierarchy level above it, such that each hierarchy
level can be seen as an interaction between an agent and an environment, even if the environment is just more agents in
a lower hierarchy level.</p>
<h2 id="agent">Agent</h2>
<p>The base agent class has 3 main function that will be used during those phases - observe, act and train.</p>
<ul>
<li><strong>Observe</strong> - this function gets the latest response from the environment as input, and updates the internal state
of the agent with the new information. The environment response will
be first passed through the agent's <strong>InputFilter</strong> object, which will process the values in the response, according
to the specific agent definition. The environment response will then be converted into a
<strong>Transition</strong> which will contain the information from a single step
(<script type="math/tex"> s_{t}, a_{t}, r_{t}, s_{t+1}, terminal signal </script>), and store it in the memory.</li>
</ul>
<p><img src="../../img/observe.png" alt="Observe" style="width: 700px;"/></p>
<ul>
<li><strong>Act</strong> - this function uses the current internal state of the agent in order to select the next action to take on
the environment. This function will call the per-agent custom function <strong>choose_action</strong> that will use the network
and the exploration policy in order to select an action. The action will be stored, together with any additional
information (like the action value for example) in an <strong>ActionInfo</strong> object. The ActionInfo object will then be
passed through the agent's <strong>OutputFilter</strong> to allow any processing of the action (like discretization,
or shifting, for example), before passing it to the environment.</li>
</ul>
<p><img src="../../img/act.png" alt="Act" style="width: 700px;"/></p>
<ul>
<li><strong>Train</strong> - this function will sample a batch from the memory and train on it. The batch of transitions will be
first wrapped into a <strong>Batch</strong> object to allow efficient querying of the batch values. It will then be passed into
the agent specific <strong>learn_from_batch</strong> function, that will extract network target values from the batch and will
train the networks accordingly. Lastly, if there's a target network defined for the agent, it will sync the target
network weights with the online network.</li>
</ul>
<p><img src="../../img/train.png" alt="Train" style="width: 700px;"/></p>
</div>
</div>
<footer>
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../network/" class="btn btn-neutral float-right" title="Network">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../features/" class="btn btn-neutral" title="Features"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
<hr/>
<div role="contentinfo">
<!-- Copyright etc -->
</div>
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
</section>
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../features/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../network/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../..';</script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../search/require.js"></script>
<script src="../../search/search.js"></script>
</body>
</html>
+328
View File
@@ -0,0 +1,328 @@
<!DOCTYPE html>
<!--[if IE 8]><html class="no-js lt-ie9" lang="en" > <![endif]-->
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="shortcut icon" href="../../img/favicon.ico">
<title>Features - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../css/highlight.css">
<link href="../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Features";
var mkdocs_page_input_path = "design/features.md";
var mkdocs_page_url = "/design/features/";
</script>
<script src="../../js/jquery-2.1.1.min.js"></script>
<script src="../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
</head>
<body class="wy-body-for-nav" role="document">
<div class="wy-grid-for-nav">
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
</form>
</div>
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<li class="toctree-l1">
<a class="" href="../..">Home</a>
</li>
<li class="toctree-l1">
<a class="" href="../../usage/">Usage</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li class=" current">
<a class="current" href="./">Features</a>
<ul class="subnav">
<li class="toctree-l3"><a href="#coach-features">Coach Features</a></li>
<ul>
<li><a class="toctree-l4" href="#supported-algorithms">Supported Algorithms</a></li>
<li><a class="toctree-l4" href="#supported-environments">Supported Environments</a></li>
</ul>
</ul>
</li>
<li class="">
<a class="" href="../control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../network/">Network</a>
</li>
<li class="">
<a class="" href="../filters/">Filters</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
&nbsp;
</nav>
<section data-toggle="wy-nav-shift" class="wy-nav-content-wrap">
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../..">Reinforcement Learning Coach</a>
</nav>
<div class="wy-nav-content">
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="../..">Docs</a> &raquo;</li>
<li>Design &raquo;</li>
<li>Features</li>
<li class="wy-breadcrumbs-aside">
</li>
</ul>
<hr/>
</div>
<div role="main">
<div class="section">
<h1 id="coach-features">Coach Features</h1>
<h2 id="supported-algorithms">Supported Algorithms</h2>
<p>Coach supports many state-of-the-art reinforcement learning algorithms, which are separated into two main classes -
value optimization and policy optimization. A detailed description of those algorithms may be found in the algorithms
section.</p>
<p style="text-align: center;">
<img src="../../img/algorithms.png" alt="Supported Algorithms" style="width: 600px;"/>
</p>
<h2 id="supported-environments">Supported Environments</h2>
<p>Coach supports a large number of environments which can be solved using reinforcement learning:</p>
<ul>
<li>
<p><strong><a href="https://github.com/deepmind/dm_control">DeepMind Control Suite</a></strong> - a set of reinforcement learning environments
powered by the MuJoCo physics engine.</p>
</li>
<li>
<p><strong><a href="https://github.com/deepmind/pysc2">Blizzard Starcraft II</a></strong> - a popular strategy game which was wrapped with a
python interface by DeepMind.</p>
</li>
<li>
<p><strong><a href="http://vizdoom.cs.put.edu.pl/">ViZDoom</a></strong> - a Doom-based AI research platform for reinforcement learning
from raw visual information.</p>
</li>
<li>
<p><strong><a href="https://github.com/carla-simulator/carla">CARLA</a></strong> - an open-source simulator for autonomous driving research.</p>
</li>
<li>
<p><strong><a href="https://gym.openai.com/">OpenAI Gym</a></strong> - a library which consists of a set of environments, from games to robotics.
Additionally, it can be extended using the API defined by the authors.</p>
</li>
</ul>
<p>In Coach, we support all the native environments in Gym, along with several extensions such as:</p>
<ul>
<li>
<p><strong><a href="https://github.com/openai/roboschool">Roboschool</a></strong> - a set of environments powered by the PyBullet engine,
that offer a free alternative to MuJoCo.</p>
</li>
<li>
<p><strong><a href="https://github.com/Breakend/gym-extensions">Gym Extensions</a></strong> - a set of environments that extends Gym for
auxiliary tasks (multitask learning, transfer learning, inverse reinforcement learning, etc.)</p>
</li>
<li>
<p><strong><a href="https://github.com/bulletphysics/bullet3/tree/master/examples/pybullet">PyBullet</a></strong> - a physics engine that
includes a set of robotics environments.</p>
</li>
</ul>
</div>
</div>
<footer>
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../control_flow/" class="btn btn-neutral float-right" title="Control Flow">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../../usage/" class="btn btn-neutral" title="Usage"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
<hr/>
<div role="contentinfo">
<!-- Copyright etc -->
</div>
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
</section>
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../../usage/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../control_flow/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../..';</script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../search/require.js"></script>
<script src="../../search/search.js"></script>
</body>
</html>
+416
View File
@@ -0,0 +1,416 @@
<!DOCTYPE html>
<!--[if IE 8]><html class="no-js lt-ie9" lang="en" > <![endif]-->
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="shortcut icon" href="../../img/favicon.ico">
<title>Filters - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../css/highlight.css">
<link href="../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Filters";
var mkdocs_page_input_path = "design/filters.md";
var mkdocs_page_url = "/design/filters/";
</script>
<script src="../../js/jquery-2.1.1.min.js"></script>
<script src="../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
</head>
<body class="wy-body-for-nav" role="document">
<div class="wy-grid-for-nav">
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
</form>
</div>
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<li class="toctree-l1">
<a class="" href="../..">Home</a>
</li>
<li class="toctree-l1">
<a class="" href="../../usage/">Usage</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li class="">
<a class="" href="../features/">Features</a>
</li>
<li class="">
<a class="" href="../control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../network/">Network</a>
</li>
<li class=" current">
<a class="current" href="./">Filters</a>
<ul class="subnav">
<li class="toctree-l3"><a href="#filters">Filters</a></li>
<ul>
<li><a class="toctree-l4" href="#input-filters">Input Filters</a></li>
<li><a class="toctree-l4" href="#output-filters">Output Filters</a></li>
</ul>
</ul>
</li>
</ul>
</li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
&nbsp;
</nav>
<section data-toggle="wy-nav-shift" class="wy-nav-content-wrap">
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../..">Reinforcement Learning Coach</a>
</nav>
<div class="wy-nav-content">
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="../..">Docs</a> &raquo;</li>
<li>Design &raquo;</li>
<li>Filters</li>
<li class="wy-breadcrumbs-aside">
</li>
</ul>
<hr/>
</div>
<div role="main">
<div class="section">
<h1 id="filters">Filters</h1>
<p>Filters are a mechanism in Coach that allows doing pre-processing and post-processing of the internal agent information.
There are two filter categories -</p>
<ul>
<li>
<p><strong>Input filters</strong> - these are filters that process the information passed <strong>into</strong> the agent from the environment.
This information includes the observation and the reward. Input filters therefore allow rescaling observations,
normalizing rewards, stack observations, etc.</p>
</li>
<li>
<p><strong>Output filters</strong> - these are filters that process the information going <strong>out</strong> of the agent into the environment.
This information includes the action the agent chooses to take. Output filters therefore allow conversion of
actions from one space into another. For example, the agent can take <script type="math/tex"> N </script> discrete actions, that will be mapped by
the output filter onto <script type="math/tex"> N </script> continuous actions.</p>
</li>
</ul>
<p>Filters can be stacked on top of each other in order to build complex processing flows of the inputs or outputs.</p>
<p style="text-align: center;">
<img src="../../img/filters.png" alt="Filters mechanism" style="width: 350px;"/>
</p>
<h2 id="input-filters">Input Filters</h2>
<p>The input filters are separated into two categories - <strong>observation filters</strong> and <strong>reward filters</strong>.</p>
<h3 id="observation-filters">Observation Filters</h3>
<ul>
<li>
<p><strong>ObservationClippingFilter</strong> - Clips the observation values to a given range of values. For example, if the
observation consists of measurements in an arbitrary range, and we want to control the minimum and maximum values
of these observations, we can define a range and clip the values of the measurements.</p>
</li>
<li>
<p><strong>ObservationCropFilter</strong> - Crops the size of the observation to a given crop window. For example, in Atari, the
observations are images with a shape of 210x160. Usually, we will want to crop the size of the observation to a
square of 160x160 before rescaling them.</p>
</li>
<li>
<p><strong>ObservationMoveAxisFilter</strong> - Reorders the axes of the observation. This can be useful when the observation is an
image, and we want to move the channel axis to be the last axis instead of the first axis.</p>
</li>
<li>
<p><strong>ObservationNormalizationFilter</strong> - Normalizes the observation values with a running mean and standard deviation of
all the observations seen so far. The normalization is performed element-wise. Additionally, when working with
multiple workers, the statistics used for the normalization operation are accumulated over all the workers.</p>
</li>
<li>
<p><strong>ObservationReductionBySubPartsNameFilter</strong> - Allows keeping only parts of the observation, by specifying their
name. For example, the CARLA environment extracts multiple measurements that can be used by the agent, such as
speed and location. If we want to only use the speed, it can be done using this filter.</p>
</li>
<li>
<p><strong>ObservationRescaleSizeByFactorFilter</strong> - Rescales an image observation by some factor. For example, the image size
can be reduced by a factor of 2.</p>
</li>
<li>
<p><strong>ObservationRescaleToSizeFilter</strong> - Rescales an image observation to a given size. The target size does not
necessarily keep the aspect ratio of the original observation.</p>
</li>
<li>
<p><strong>ObservationRGBToYFilter</strong> - Converts a color image observation specified using the RGB encoding into a grayscale
image observation, by keeping only the luminance (Y) channel of the YUV encoding. This can be useful if the colors
in the original image are not relevant for solving the task at hand.</p>
</li>
<li>
<p><strong>ObservationSqueezeFilter</strong> - Removes redundant axes from the observation, which are axes with a dimension of 1.</p>
</li>
<li>
<p><strong>ObservationStackingFilter</strong> - Stacks several observations on top of each other. For image observation this will
create a 3D blob. The stacking is done in a lazy manner in order to reduce memory consumption. To achieve this,
a LazyStack object is used in order to wrap the observations in the stack. For this reason, the
ObservationStackingFilter <strong>must</strong> be the last filter in the inputs filters stack.</p>
</li>
<li>
<p><strong>ObservationUint8Filter</strong> - Converts a floating point observation into an unsigned int 8 bit observation. This is
mostly useful for reducing memory consumption and is usually used for image observations. The filter will first
spread the observation values over the range 0-255 and then discretize them into integer values.</p>
</li>
</ul>
<h3 id="reward-filters">Reward Filters</h3>
<ul>
<li>
<p><strong>RewardClippingFilter</strong> - Clips the reward values into a given range. For example, in DQN, the Atari rewards are
clipped into the range -1 and 1 in order to control the scale of the returns.</p>
</li>
<li>
<p><strong>RewardNormalizationFilter</strong> - Normalizes the reward values with a running mean and standard deviation of
all the rewards seen so far. When working with multiple workers, the statistics used for the normalization operation
are accumulated over all the workers.</p>
</li>
<li>
<p><strong>RewardRescaleFilter</strong> - Rescales the reward by a given factor. Rescaling the rewards of the environment has been
observed to have a large effect (negative or positive) on the behavior of the learning process.</p>
</li>
</ul>
<h2 id="output-filters">Output Filters</h2>
<p>The output filters only process the actions.</p>
<h3 id="action-filters">Action Filters</h3>
<ul>
<li>
<p><strong>AttentionDiscretization</strong> - Discretizes an <strong>AttentionActionSpace</strong>. The attention action space defines the actions
as choosing sub-boxes in a given box. For example, consider an image of size 100x100, where the action is choosing
a crop window of size 20x20 to attend to in the image. AttentionDiscretization allows discretizing the possible crop
windows to choose into a finite number of options, and map a discrete action space into those crop windows.</p>
</li>
<li>
<p><strong>BoxDiscretization</strong> - Discretizes a continuous action space into a discrete action space, allowing the usage of
agents such as DQN for continuous environments such as MuJoCo. Given the number of bins to discretize into, the
original continuous action space is uniformly separated into the given number of bins, each mapped to a discrete
action index. For example, if the original actions space is between -1 and 1 and 5 bins were selected, the new action
space will consist of 5 actions mapped to -1, -0.5, 0, 0.5 and 1.</p>
</li>
<li>
<p><strong>BoxMasking</strong> - Masks part of the action space to enforce the agent to work in a defined space. For example,
if the original action space is between -1 and 1, then this filter can be used in order to constrain the agent actions
to the range 0 and 1 instead. This essentially masks the range -1 and 0 from the agent.</p>
</li>
<li>
<p><strong>PartialDiscreteActionSpaceMap</strong> - Partial map of two countable action spaces. For example, consider an environment
with a MultiSelect action space (select multiple actions at the same time, such as jump and go right), with 8 actual
MultiSelect actions. If we want the agent to be able to select only 5 of those actions by their index (0-4), we can
map a discrete action space with 5 actions into the 5 selected MultiSelect actions. This will both allow the agent to
use regular discrete actions, and mask 3 of the actions from the agent.</p>
</li>
<li>
<p><strong>FullDiscreteActionSpaceMap</strong> - Full map of two countable action spaces. This works in a similar way to the
PartialDiscreteActionSpaceMap, but maps the entire source action space into the entire target action space, without
masking any actions.</p>
</li>
<li>
<p><strong>LinearBoxToBoxMap</strong> - A linear mapping of two box action spaces. For example, if the action space of the
environment consists of continuous actions between 0 and 1, and we want the agent to choose actions between -1 and 1,
the LinearBoxToBoxMap can be used to map the range -1 and 1 to the range 0 and 1 in a linear way. This means that the
action -1 will be mapped to 0, the action 1 will be mapped to 1, and the rest of the actions will be linearly mapped
between those values.</p>
</li>
</ul>
</div>
</div>
<footer>
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../../algorithms/value_optimization/dqn/" class="btn btn-neutral float-right" title="DQN">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../network/" class="btn btn-neutral" title="Network"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
<hr/>
<div role="contentinfo">
<!-- Copyright etc -->
</div>
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
</section>
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../network/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../../algorithms/value_optimization/dqn/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../..';</script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../search/require.js"></script>
<script src="../../search/search.js"></script>
</body>
</html>
-363
View File
@@ -1,363 +0,0 @@
<!DOCTYPE html>
<!--[if IE 8]><html class="no-js lt-ie9" lang="en" > <![endif]-->
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Design - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../img/favicon.ico">
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../css/highlight.css">
<link href="../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Design";
</script>
<script src="../js/jquery-2.1.1.min.js"></script>
<script src="../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../js/highlight.pack.js"></script>
<script src="../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
</head>
<body class="wy-body-for-nav" role="document">
<div class="wy-grid-for-nav">
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../index.html" class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
</form>
</div>
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../index.html">Home</a>
</li>
<li>
<li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Design</a>
<ul>
<li class="toctree-l3"><a href="#coach-design">Coach Design</a></li>
<li><a class="toctree-l4" href="#network-design">Network Design</a></li>
<li><a class="toctree-l4" href="#keeping-network-copies-in-sync">Keeping Network Copies in Sync</a></li>
<li><a class="toctree-l4" href="#supported-algorithms">Supported Algorithms</a></li>
</ul>
</li>
<li>
<li>
<li class="toctree-l1 ">
<a class="" href="../usage/index.html">Usage</a>
</li>
<li>
<li>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/imitation/bc/index.html">Behavioral Cloning</a>
</li>
</ul>
<li>
<li>
<li class="toctree-l1 ">
<a class="" href="../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../contributing/add_agent/index.html">Adding a New Agent</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../contributing/add_env/index.html">Adding a New Environment</a>
</li>
</ul>
<li>
</ul>
</div>
&nbsp;
</nav>
<section data-toggle="wy-nav-shift" class="wy-nav-content-wrap">
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../index.html">Reinforcement Learning Coach Documentation</a>
</nav>
<div class="wy-nav-content">
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="../index.html">Docs</a> &raquo;</li>
<li>Design</li>
<li class="wy-breadcrumbs-aside">
</li>
</ul>
<hr/>
</div>
<div role="main">
<div class="section">
<h1 id="coach-design">Coach Design</h1>
<h2 id="network-design">Network Design</h2>
<p>Each agent has at least one neural network, used as the function approximator, for choosing the actions. The network is designed in a modular way to allow reusability in different agents. It is separated into three main parts:</p>
<ul>
<li>
<p><strong>Input Embedders</strong> - This is the first stage of the network, meant to convert the input into a feature vector representation. It is possible to combine several instances of any of the supported embedders, in order to allow varied combinations of inputs. </p>
<p>There are two main types of input embedders: </p>
<ol>
<li>Image embedder - Convolutional neural network. </li>
<li>Vector embedder - Multi-layer perceptron. </li>
</ol>
</li>
<li>
<p><strong>Middlewares</strong> - The middleware gets the output of the input embedder, and processes it into a different representation domain, before sending it through the output head. The goal of the middleware is to enable processing the combined outputs of several input embedders, and pass them through some extra processing. This, for instance, might include an LSTM or just a plain simple FC layer.</p>
</li>
<li>
<p><strong>Output Heads</strong> - The output head is used in order to predict the values required from the network. These might include action-values, state-values or a policy. As with the input embedders, it is possible to use several output heads in the same network. For example, the <em>Actor Critic</em> agent combines two heads - a policy head and a state-value head.
In addition, the output heads defines the loss function according to the head type.</p>
</li>
</ul>
<p>​</p>
<p style="text-align: center;">
<img src="../img/network.png" alt="Network Design" style="width: 400px;"/>
</p>
<h2 id="keeping-network-copies-in-sync">Keeping Network Copies in Sync</h2>
<p>Most of the reinforcement learning agents include more than one copy of the neural network. These copies serve as counterparts of the main network which are updated in different rates, and are often synchronized either locally or between parallel workers. For easier synchronization of those copies, a wrapper around these copies exposes a simplified API, which allows hiding these complexities from the agent. </p>
<p style="text-align: center;">
<img src="../img/distributed.png" alt="Distributed Training" style="width: 600px;"/>
</p>
<h2 id="supported-algorithms">Supported Algorithms</h2>
<p>Coach supports many state-of-the-art reinforcement learning algorithms, which are separated into two main classes - value optimization and policy optimization. A detailed description of those algorithms may be found in the algorithms section.</p>
<p style="text-align: center;">
<img src="../img/algorithms.png" alt="Supported Algorithms" style="width: 600px;"/>
</p>
</div>
</div>
<footer>
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../usage/index.html" class="btn btn-neutral float-right" title="Usage"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../index.html" class="btn btn-neutral" title="Home"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
<hr/>
<div role="contentinfo">
<!-- Copyright etc -->
</div>
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
</section>
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../usage/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
</body>
</html>
+310
View File
@@ -0,0 +1,310 @@
<!DOCTYPE html>
<!--[if IE 8]><html class="no-js lt-ie9" lang="en" > <![endif]-->
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="shortcut icon" href="../../img/favicon.ico">
<title>Network - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../../css/highlight.css">
<link href="../../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Network";
var mkdocs_page_input_path = "design/network.md";
var mkdocs_page_url = "/design/network/";
</script>
<script src="../../js/jquery-2.1.1.min.js"></script>
<script src="../../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../../js/highlight.pack.js"></script>
</head>
<body class="wy-body-for-nav" role="document">
<div class="wy-grid-for-nav">
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../.." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
</form>
</div>
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<li class="toctree-l1">
<a class="" href="../..">Home</a>
</li>
<li class="toctree-l1">
<a class="" href="../../usage/">Usage</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li class="">
<a class="" href="../features/">Features</a>
</li>
<li class="">
<a class="" href="../control_flow/">Control Flow</a>
</li>
<li class=" current">
<a class="current" href="./">Network</a>
<ul class="subnav">
<li class="toctree-l3"><a href="#network-design">Network Design</a></li>
<ul>
<li><a class="toctree-l4" href="#keeping-network-copies-in-sync">Keeping Network Copies in Sync</a></li>
</ul>
</ul>
</li>
<li class="">
<a class="" href="../filters/">Filters</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
&nbsp;
</nav>
<section data-toggle="wy-nav-shift" class="wy-nav-content-wrap">
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../..">Reinforcement Learning Coach</a>
</nav>
<div class="wy-nav-content">
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="../..">Docs</a> &raquo;</li>
<li>Design &raquo;</li>
<li>Network</li>
<li class="wy-breadcrumbs-aside">
</li>
</ul>
<hr/>
</div>
<div role="main">
<div class="section">
<h1 id="network-design">Network Design</h1>
<p>Each agent has at least one neural network, used as the function approximator, for choosing the actions. The network is designed in a modular way to allow reusability in different agents. It is separated into three main parts:</p>
<ul>
<li>
<p><strong>Input Embedders</strong> - This is the first stage of the network, meant to convert the input into a feature vector representation. It is possible to combine several instances of any of the supported embedders, in order to allow varied combinations of inputs. </p>
<p>There are two main types of input embedders: </p>
<ol>
<li>Image embedder - Convolutional neural network. </li>
<li>Vector embedder - Multi-layer perceptron. </li>
</ol>
</li>
<li>
<p><strong>Middlewares</strong> - The middleware gets the output of the input embedder, and processes it into a different representation domain, before sending it through the output head. The goal of the middleware is to enable processing the combined outputs of several input embedders, and pass them through some extra processing. This, for instance, might include an LSTM or just a plain simple FC layer.</p>
</li>
<li>
<p><strong>Output Heads</strong> - The output head is used in order to predict the values required from the network. These might include action-values, state-values or a policy. As with the input embedders, it is possible to use several output heads in the same network. For example, the <em>Actor Critic</em> agent combines two heads - a policy head and a state-value head.
In addition, the output heads defines the loss function according to the head type.</p>
</li>
</ul>
<p>​</p>
<p style="text-align: center;">
<img src="../../img/network.png" alt="Network Design" style="width: 400px;"/>
</p>
<h2 id="keeping-network-copies-in-sync">Keeping Network Copies in Sync</h2>
<p>Most of the reinforcement learning agents include more than one copy of the neural network. These copies serve as counterparts of the main network which are updated in different rates, and are often synchronized either locally or between parallel workers. For easier synchronization of those copies, a wrapper around these copies exposes a simplified API, which allows hiding these complexities from the agent. </p>
<p style="text-align: center;">
<img src="../../img/distributed.png" alt="Distributed Training" style="width: 600px;"/>
</p>
</div>
</div>
<footer>
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../filters/" class="btn btn-neutral float-right" title="Filters">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../control_flow/" class="btn btn-neutral" title="Control Flow"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
<hr/>
<div role="contentinfo">
<!-- Copyright etc -->
</div>
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
</section>
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../control_flow/" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../filters/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '../..';</script>
<script src="../../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../../search/require.js"></script>
<script src="../../search/search.js"></script>
</body>
</html>
+1
View File
@@ -0,0 +1 @@
<mxfile userAgent="Mozilla/5.0 (Macintosh; Intel Mac OS X 10_13_4) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/67.0.3396.99 Safari/537.36" version="9.0.0" editor="www.draw.io" type="device"><diagram id="33c2a640-8c1e-935c-0e0a-86b5dd5c932c" name="Page-1">7V1td5u4Ev41+dgehHj9mKRNe89pu93tuS/7kdjEZheDL8ZJc3/9lTDCoFGMbEsYXGX37NoyYMwzM5p5ZjS6wfern5+KaL38ms/j9Ma25j9v8Icb27ZxYJH/0ZHX3QiyPW83siiSeT22H/iR/C+uB+sTF9tkHm86B5Z5npbJujs4y7MsnpWdsago8pfuYU952v3WdbSIwcCPWZTC0X8n83K5Gw1caz/+OU4WS/bNyKo/eYxmfy+KfJvV33dj46fqb/fxKmLXqo/fLKN5/tIawh9v8H2R5+Xu1ernfZzSh8se2+68hzc+be67iLNS6gSMd6c8R+k2Zvdc3Vn5yp5G9Xtiega6wXcvy6SMf6yjGf30hQgAGVuWq7T+eJFGG/r0LfJ6lq+SWf16Uxb53/F9nuZFdVXszYL48an5hD1nTEaekjRtHTmP4uBpRsfzrKyFxbbq963jrOqPjEdpssjIWBo/lfRtMavP8sg7+Ijqp/YcF2X8szVUP7JPcb6Ky+KVHMI+dZiEMwG3wmA38LIXF4cJxbIlKtirn29Ui+iiufoeJvKiRkqMmhdoBG0ebZbNeQcR/OjRf2QQrHWAfFERzZN4j1qWZ7EKYH01wBJ8OrjalmdBXB0Brh6zDefgGloG11IRkn5XQ6mNBkhiEZLIUYEkAkhGZJ4axrKG/gfL9xXjh+wD+KkALHTeuxxkXvg+wPs/FwDYaEwbQCX42f2aGGfzW+prkHczCg0FpI1XM7fTJ5RGj3F613gHrYd4V/0jRkvy8ZOnXrz+h34TeYT12z/rL34TmjIqFnHZFdd43vGLIFitZ+8KHj0bK+I0KpPnrjclwqP+hu95Qu6ukYXGwjJJaBxDdo1Nvi1mcX1a278BV3J5qcIWd63dgwDXIthGr63D1vSAzTE3ja2D9+aG/pknIPfwCeCWuBPIi92v3OtEg7WUmvgSamImLMkJi3M9sOyEZfMCfYrBQzpdyuHBO2guO4KlPDiwuzrXBI0tGD1PFBmEKuYtg6IWFFEoiaLDz1QnKaP3y8OoKKDD/Awo0EaMRUZVhRcJZ8ftJiYDWVy+5MXfFatFf2ERzxMaHVgkRkjyjLyYJwSA5HG7ezvWqGHA8AB7XSBFs6O2aAByZP+scCyX9L/xzzX5yVGN3DpPk9krg3YTrdYpPeipyFfNGTXgBEKDdD/S/oBIOwDpWRFHJQUtytoKmmRPFN/88a94p7kZBXOn3knZwF/mxV5QmpN3B0dZFV7M5wkdjtL9VaPHfFuC04xsANlw8ICyASflRrsjGvMSVCsdb2DeZslTXqzSV17/myM2FWD2PR1L82xR/RLyQ8gn5P6odYifom3amRpqGaluY2OkogXuJQwGDJv2BoPiTaRikzB52CQVxDshIM++ZCZiXr1gdmbeHJPFL3SCyZ6pfMWbdZ5RAzNWyEdAIlpdsid0BhQF5pH0cL7kp5QcUdh59uyptR5yPcRIhxl5QDEZv6MPJplF6W39wSqZz+nXCIVgLyaWLNEiy4ZUx9W/UJCCPFehbYGvJyZCFID4axG/9Zw2Et4Xu13gkcchKk37WtyF+KhcEefL33Afg4sCQEZzwnkeH4sksvlXJL32dUovCjkp4ZnBC0mva2mWXhh9Fdsqpl5GVWBVjVGfKskqT8oj/jeRlexxs25EZvIz7UGVUT3TYj5NbvlgqkVCfkzFXOtK4b0k/vR2bdBW4B4jCbRtXWj7AO3ZMs83ByiO8yGt83PX6ijzfIgtCH0b30c5oDDZlD9u4uIZhqjT1M1LxjyBAMlAE5DsGhPxGn8mZctpJO/+rO/hJH9y55TdtKOhtovJ6mNH4mPyAYTju6f5mLzv5/GCpMjH5G/YRYd9TP6+uOPP9jEbxfzVZR1NQNi5OMjhWbpThd3HeoSdD5AuLuyQzOLSCFcxTw8bMbkhV1HoCehJZpaVT9W/FsHDSn5HYpB4fQ1OLUvF3PqR8Oya1I4AHWklIO3Sk3EyZkLCTHCiYkMjIUpEKTESkFfpFiLs0oos4ZhkzQezbVH95v2Hz1GRRI+pyToekXVElgXB1pZ2tCGvEs3nDaTEQBRJRicBqtAvUVF9lDeftzPUBmJZnnS/4rMNsSgpqQRiieLsic/67dCE2a9OqscftSeArNNjE+e91/rzna5nMFDmEllBz1qS8PAJZwcr+Eoi86NEfLTi7J8sztyVhkpe9gown/w4/gSwmL73t9uKVQRGf2TKpI40qCzk61O7tWf5tlxXFabs9LHO/Bf04twha8cwjMGu1/iNjGZ0ETcDd8VAlSX0ziYdz4jxMQwJOcvBh/ycxUiyaRiMy4cKvB0JgiHtiCDJ3orp4/VNHe/H62STz6siC1ZYsditZGWBjeF7juN7eA7A9QDsSBctjGGA2GV8OnXoLcJnFa/y+quMOktG/oKWMNoifyyxoPd6/IJgVH6BwyNvn1rQOVhQBIL0i4csTAlaAjyvdPjBzDIKZhnRemZfoBIqJhnnYq3iRrMqXVNzAdEyRnF3ASVrXGGsycUCe29hAwKBddFdwW4CgqPTQy6+WF8rB8aBVTn9lRjjS1ZsIsfpg1VXAacj0S9kRF7iuVUhrj0qPxH4UI4lFIRzi0SQEx687pB8kiNY95xnVHfemETId1adDW5oj9xyRpe4M27azBzyM4dYAoaYOZjSidofpEm1lD1/ArB32h4YsuFowB3/coBDKnEg6Ebj7itqQsXXwCNPQCEJu1CpaAnnQgopjaMi22tnbZCVe4BXvviqYcobWAVMP7I1Ub4eJFa26/muaK92llphmjG50ibXv1zXYW9SJTBvPvzOkqr6sYzFV3fCLtqnl3DzVwp5CRjSH2e7CkxDcnQu4atBGHW8SAyM5wXI8VwLB7xzECAl0SPfAmW/lYHuQqzAOZxzcL1zT3B6+l7DiFxx32sP0mhpntPk+lNOadRsu7ppN6qoUu80HiqTlaBp2zTJtkEzH0AGPAFnrmtBhScKhbyUYjZPnjtYev/d0r177iik72p0bmmITH3h5lPyqq622F1ls6bdJXdjaU5QqMfJjbU/ag1X38tGjTgdK0584wQUiBo8axInH7ruZnOQg3DZoQ+2cbicm+5DN12cRDEIHkLQsS+HoM5U9i++b4UrmJj1bbTkw2S2saXHaqIbXE4ToScd09cVkXUVfs2wnKQA3N5qA11paR+mpet5knVAN+1xTwOZC+2xYHG5tq59vg8wo8QK+4E1GF2D235qLWT3PJDfJoK4IgEfMEHHM1VRUTLuq77BauwhoT+z/kIBO0YG60Po7f4Vl+VrDXi0LXMylBflMl/Q7v1faPzfFtg3iLGDkvKmcLQpLjbhtikuX3ZJjTR3Jd0AHQYzUD7GwzyeW3vC/I2RcImOw3lfYJWR/E55AT9z+Jp6gsGb7uHzPJ7LPPoEtorr7V/Pe7HdE85mDINpNe7RR9D7IqUaV5LIcTlhaLZ7Plqp+Ct5mpbzA/llKwGkNeToE3CPDoJbwopJ+AC6l8lqXeSm/+jNKU4lv+mC1Uuw6YobgitZ9HS+rRR5eva4bCUvNt6p66V9vhs9uJI296PP9IHfeOwJdl9zFf6WbMXNVUKR+6Erg9U0bL9oDuvaq88E+6UPSQOEEh0rJlQeegxrvi8QDdRQssjiV4RZGK4ed0Sd+R3WOeQsLCEJS57Ymhbpr6IsWgjWd43PoxqDUu7bgB1sAyCaj5Xo5MX2DfZmQfz4JKOT8ygOnmYqlA8pSogQjDjQkGjrdVE1AOLZlpNQg2tp0vg5TrVqn9rZcRy6Bzo1CDKR+nRPImjRo3uB/Yg9qflw7sbB3Dld99SgxNU7IkHVfCAommetMM7bolFUYa3LDyaqS538a3KDx6HofJcLxBiDAYq3yBRvFF0GJQRRGlLRoUtrFH1qig7W5gyr6Bfzpqel6BhstupDlPQp+pC12UbR9Sg6RkDRgwEVXaK9ugmb+8NmW7SsWVvYTNxAAJuJm1XEzbZoHbOuuLnhWsw02wcTZyOxgN7QNs0iEzhfgabzIRnGA/rTyDaKflLgPKyiD5kpNoo+UOA8rKIbhkwKJRA4Y3vAwBkZhmz6ig4CZ4wHDJzRxRiy0PNxJKXoMSKq7l94Rnf52nrkCFLMiFULq9f1IUmyOHtOijxbGY3XofHIx1CW4LSB2FSiXukNW3YKbi4/2dsORE3MlqnoAoiQYcuU6J/Ll+rZnkD7tLFltmHLJGHiHSOBkdTmW9uGLbsCTedrObE9oG/N9Noo+mGU+LUWAyu6Ycumr+gev6vOsIpu2DIplPj1YUTRIUr6FN2wZdNXdN8Diu4PqOgXC5ynxZZ5FmA4sLAiQRdbZsNQ2bBl09R4DwHmFQtqSPWxZRjG64B2sZ6T+MXgKWcbOCfAEfRtRaGvC00Y14OFkgZNeS4bI8BlD4sn9L0NKaqCFEUC31wfKSrY3X4gz2paIRQgRZE3YAiFRdliE0JNTNN5ug35A4ZQ+GJ7kk1L0QEpOqyiiwJdo+jTUnRAig6r6GY1vRRKgBRFHkRJm6I7omS0UfRpKTogRRHr2z2EojuiTPkgit48Ufk2UrItAfflQ54YMQCPAMQDYTO/MbRvwaAZC6Mtfg+w0zCzAWYf2yzmIPANv0msZlSBHvqsLVeHChHYcjXLKR2Yyr5diPC82tZf5+GHIdXsC4p8fUHWQo0lhSRIkq23tMskgaIUUFmTdaAa+VFiTUMOtVBUmi3axlfF/ifIgSFtvi0Nbv36FgLyeGDkJPLine truncated
Binary file not shown.
File diff suppressed because it is too large. Load diff

Before

Width:  |  Height:  |  Size: 355 KiB

After

Width:  |  Height:  |  Size: 193 KiB

Binary file not shown.
Binary file not shown.
BIN
View File
Binary file not shown.

After

Width:  |  Height:  |  Size: 49 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 21 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 29 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 32 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 24 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 40 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 39 KiB

+158 -202
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<meta name="description" content="Reinforcement Learning Coach by Intel Nervana.">
<title>Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="./img/favicon.ico">
<title>Home - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="./css/theme.css" type="text/css" />
<link rel="stylesheet" href="./css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="./css/highlight.css">
<link href="./extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "None";
var mkdocs_page_name = "Home";
var mkdocs_page_input_path = "index.md";
var mkdocs_page_url = "/";
</script>
<script src="./js/jquery-2.1.1.min.js"></script>
<script src="./js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="./js/highlight.pack.js"></script>
<script src="./js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="./js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="./index.html" class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="./search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,197 +45,152 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Home</a>
<ul>
<li class="toctree-l3"><a href="#what-is-coach">What is Coach?</a></li>
<li><a class="toctree-l4" href="#motivation">Motivation</a></li>
<li><a class="toctree-l4" href="#solution">Solution</a></li>
<li><a class="toctree-l4" href="#design">Design</a></li>
</ul>
</li>
<li>
<li>
<li class="toctree-l1 ">
<a class="" href="design/index.html">Design</a>
</li>
<li>
<li>
<li class="toctree-l1 ">
<a class="" href="usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1 current">
<a class="current" href=".">Home</a>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/dqn/index.html">DQN</a>
<li class="toctree-l2"><a href="#what-is-coach">What is Coach?</a></li>
<ul>
</li>
<li><a class="toctree-l3" href="#motivation">Motivation</a></li>
<li><a class="toctree-l3" href="#solution">Solution</a></li>
<li><a class="toctree-l3" href="#design">Design</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/imitation/bc/index.html">Behavioral Cloning</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="usage/">Usage</a>
</li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="contributing/add_agent/index.html">Adding a New Agent</a>
</li>
<li class="toctree-l1 ">
<a class="" href="contributing/add_env/index.html">Adding a New Environment</a>
</li>
<li class="">
<a class="" href="design/features/">Features</a>
</li>
<li class="">
<a class="" href="design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="design/network/">Network</a>
</li>
<li class="">
<a class="" href="design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -251,7 +202,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="./index.html">Reinforcement Learning Coach Documentation</a>
<a href=".">Reinforcement Learning Coach</a>
</nav>
@@ -259,7 +210,7 @@
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="./index.html">Docs</a> &raquo;</li>
<li><a href=".">Docs</a> &raquo;</li>
@@ -281,7 +232,7 @@
With Coach, it is possible to model an agent by combining various building blocks, and training the agent on multiple environments.
The available environments allow testing the agent in different practical fields such as robotics, autonomous driving, games and more.
Coach collects statistics from the training process and supports advanced visualization techniques for debugging the agent being trained.</p>
<p>Blog post from the Intel® Nervana™ website can be found <a href="https://www.intelnervana.com/reinforcement-learning-coach-intel">here</a>. </p>
<p>Blog post from the Intel® AI website can be found <a href="https://ai.intel.com/reinforcement-learning-coach-intel/">here</a>.</p>
<p>GitHub repository is <a href="https://github.com/NervanaSystems/coach">here</a>. </p>
<h2 id="design">Design</h2>
<p><img src="img/design.png" alt="Coach Design" style="width: 800px;"/></p>
@@ -292,7 +243,7 @@ Coach collects statistics from the training process and supports advanced visual
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="design/index.html" class="btn btn-neutral float-right" title="Design"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="usage/" class="btn btn-neutral float-right" title="Usage">Next <span class="icon icon-circle-arrow-right"></span></a>
</div>
@@ -307,7 +258,7 @@ Coach collects statistics from the training process and supports advanced visual
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -315,20 +266,25 @@ Coach collects statistics from the training process and supports advanced visual
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span style="margin-left: 15px"><a href="design/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="usage/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '.';</script>
<script src="./js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="./search/require.js"></script>
<script src="./search/search.js"></script>
</body>
</html>
<!--
MkDocs version : 0.14.0
Build Date UTC : 2017-12-18 18:59:45.506407
MkDocs version : 0.17.5
Build Date UTC : 2018-08-09 12:14:19
-->
+45 -1
View File
@@ -1,5 +1,4 @@
$( document ).ready(function() {
// Shift nav in mobile when clicking the menu.
$(document).on('click', "[data-toggle='wy-nav-top']", function() {
$("[data-toggle='wy-nav-shift']").toggleClass("shift");
@@ -12,6 +11,23 @@ $( document ).ready(function() {
$("[data-toggle='rst-versions']").toggleClass("shift");
});
// Keyboard navigation
document.addEventListener("keydown", function(e) {
if ($(e.target).is(':input')) return true;
var key = e.which || e.keyCode || window.event && window.event.keyCode;
var page;
switch (key) {
case 39: // right arrow
page = $('[role="navigation"] a:contains(Next):first').prop('href');
break;
case 37: // left arrow
page = $('[role="navigation"] a:contains(Previous):first').prop('href');
break;
default: break;
}
if (page) window.location.href = page;
});
$(document).on('click', "[data-toggle='rst-current-version']", function() {
$("[data-toggle='rst-versions']").toggleClass("shift-up");
});
@@ -53,3 +69,31 @@ window.SphinxRtdTheme = (function (jquery) {
StickyNav : stickyNav
};
}($));
// The code below is a copy of @seanmadsen code posted Jan 10, 2017 on issue 803.
// https://github.com/mkdocs/mkdocs/issues/803
// This just incorporates the auto scroll into the theme itself without
// the need for additional custom.js file.
//
$(function() {
$.fn.isFullyWithinViewport = function(){
var viewport = {};
viewport.top = $(window).scrollTop();
viewport.bottom = viewport.top + $(window).height();
var bounds = {};
bounds.top = this.offset().top;
bounds.bottom = bounds.top + this.outerHeight();
return ( ! (
(bounds.top <= viewport.top) ||
(bounds.bottom >= viewport.bottom)
) );
};
if( $('li.toctree-l1.current').length && !$('li.toctree-l1.current').isFullyWithinViewport() ) {
$('.wy-nav-side')
.scrollTop(
$('li.toctree-l1.current').offset().top -
$('.wy-nav-side').offset().top -
60
);
}
});
-7
View File
@@ -1,7 +0,0 @@
/**
* lunr - http://lunrjs.com - A bit like Solr, but much smaller and not as bright - 0.5.7
* Copyright (C) 2014 Oliver Nightingale
* MIT Licensed
* @license
*/
!function(){var t=function(e){var n=new t.Index;return n.pipeline.add(t.trimmer,t.stopWordFilter,t.stemmer),e&&e.call(n,n),n};t.version="0.5.7",t.utils={},t.utils.warn=function(t){return function(e){t.console&&console.warn&&console.warn(e)}}(this),t.EventEmitter=function(){this.events={}},t.EventEmitter.prototype.addListener=function(){var t=Array.prototype.slice.call(arguments),e=t.pop(),n=t;if("function"!=typeof e)throw new TypeError("last argument must be a function");n.forEach(function(t){this.hasHandler(t)||(this.events[t]=[]),this.events[t].push(e)},this)},t.EventEmitter.prototype.removeListener=function(t,e){if(this.hasHandler(t)){var n=this.events[t].indexOf(e);this.events[t].splice(n,1),this.events[t].length||delete this.events[t]}},t.EventEmitter.prototype.emit=function(t){if(this.hasHandler(t)){var e=Array.prototype.slice.call(arguments,1);this.events[t].forEach(function(t){t.apply(void 0,e)})}},t.EventEmitter.prototype.hasHandler=function(t){return t in this.events},t.tokenizer=function(t){if(!arguments.length||null==t||void 0==t)return[];if(Array.isArray(t))return t.map(function(t){return t.toLowerCase()});for(var e=t.toString().replace(/^\s+/,""),n=e.length-1;n>=0;n--)if(/\S/.test(e.charAt(n))){e=e.substring(0,n+1);break}return e.split(/(?:\s+|\-)/).filter(function(t){return!!t}).map(function(t){return t.toLowerCase()})},t.Pipeline=function(){this._stack=[]},t.Pipeline.registeredFunctions={},t.Pipeline.registerFunction=function(e,n){n in this.registeredFunctions&&t.utils.warn("Overwriting existing registered function: "+n),e.label=n,t.Pipeline.registeredFunctions[e.label]=e},t.Pipeline.warnIfFunctionNotRegistered=function(e){var n=e.label&&e.label in this.registeredFunctions;n||t.utils.warn("Function is not registered with pipeline. This may cause problems when serialising the index.\n",e)},t.Pipeline.load=function(e){var n=new t.Pipeline;return e.forEach(function(e){var i=t.Pipeline.registeredFunctions[e];if(!i)throw new Error("Cannot load un-registered function: "+e);n.add(i)}),n},t.Pipeline.prototype.add=function(){var e=Array.prototype.slice.call(arguments);e.forEach(function(e){t.Pipeline.warnIfFunctionNotRegistered(e),this._stack.push(e)},this)},t.Pipeline.prototype.after=function(e,n){t.Pipeline.warnIfFunctionNotRegistered(n);var i=this._stack.indexOf(e)+1;this._stack.splice(i,0,n)},t.Pipeline.prototype.before=function(e,n){t.Pipeline.warnIfFunctionNotRegistered(n);var i=this._stack.indexOf(e);this._stack.splice(i,0,n)},t.Pipeline.prototype.remove=function(t){var e=this._stack.indexOf(t);this._stack.splice(e,1)},t.Pipeline.prototype.run=function(t){for(var e=[],n=t.length,i=this._stack.length,o=0;n>o;o++){for(var r=t[o],s=0;i>s&&(r=this._stack[s](r,o,t),void 0!==r);s++);void 0!==r&&e.push(r)}return e},t.Pipeline.prototype.reset=function(){this._stack=[]},t.Pipeline.prototype.toJSON=function(){return this._stack.map(function(e){return t.Pipeline.warnIfFunctionNotRegistered(e),e.label})},t.Vector=function(){this._magnitude=null,this.list=void 0,this.length=0},t.Vector.Node=function(t,e,n){this.idx=t,this.val=e,this.next=n},t.Vector.prototype.insert=function(e,n){var i=this.list;if(!i)return this.list=new t.Vector.Node(e,n,i),this.length++;for(var o=i,r=i.next;void 0!=r;){if(e<r.idx)return o.next=new t.Vector.Node(e,n,r),this.length++;o=r,r=r.next}return o.next=new t.Vector.Node(e,n,r),this.length++},t.Vector.prototype.magnitude=function(){if(this._magniture)return this._magnitude;for(var t,e=this.list,n=0;e;)t=e.val,n+=t*t,e=e.next;return this._magnitude=Math.sqrt(n)},t.Vector.prototype.dot=function(t){for(var e=this.list,n=t.list,i=0;e&&n;)e.idx<n.idx?e=e.next:e.idx>n.idx?n=n.next:(i+=e.val*n.val,e=e.next,n=n.next);return i},t.Vector.prototype.similarity=function(t){return this.dot(t)/(this.magnitude()*t.magnitude())},t.SortedSet=function(){this.length=0,this.elements=[]},t.SortedSet.load=function(t){var e=new this;return e.elements=t,e.length=t.length,e},t.SortedSet.prototype.add=function(){Array.prototype.slice.call(arguments).forEach(function(t){~this.indexOf(t)||this.elements.splice(this.locationFor(t),0,t)},this),this.length=this.elements.length},t.SortedSet.prototype.toArray=function(){return this.elements.slice()},t.SortedSet.prototype.map=function(t,e){return this.elements.map(t,e)},t.SortedSet.prototype.forEach=function(t,e){return this.elements.forEach(t,e)},t.SortedSet.prototype.indexOf=function(t,e,n){var e=e||0,n=n||this.elements.length,i=n-e,o=e+Math.floor(i/2),r=this.elements[o];return 1>=i?r===t?o:-1:t>r?this.indexOf(t,o,n):r>t?this.indexOf(t,e,o):r===t?o:void 0},t.SortedSet.prototype.locationFor=function(t,e,n){var e=e||0,n=n||this.elements.length,i=n-e,o=e+Math.floor(i/2),r=this.elements[o];if(1>=i){if(r>t)return o;if(t>r)return o+1}return t>r?this.locationFor(t,o,n):r>t?this.locationFor(t,e,o):void 0},t.SortedSet.prototype.intersect=function(e){for(var n=new t.SortedSet,i=0,o=0,r=this.length,s=e.length,a=this.elements,h=e.elements;;){if(i>r-1||o>s-1)break;a[i]!==h[o]?Line truncated
-454
View File
@@ -1,454 +0,0 @@
{
"docs": [
{
"location": "/index.html",
"text": "What is Coach?\n\n\nMotivation\n\n\nTrain and evaluate reinforcement learning agents by harnessing the power of multi-core CPU processing to achieve state-of-the-art results. Provide a sandbox for easing the development process of new algorithms through a modular design and an elegant set of APIs. \n\n\nSolution\n\n\nCoach is a python environment which models the interaction between an agent and an environment in a modular way.\nWith Coach, it is possible to model an agent by combining various building blocks, and training the agent on multiple environments.\nThe available environments allow testing the agent in different practical fields such as robotics, autonomous driving, games and more. \nCoach collects statistics from the training process and supports advanced visualization techniques for debugging the agent being trained.\n\n\nBlog post from the Intel\u00ae Nervana\u2122 website can be found \nhere\n. \n\n\nGitHub repository is \nhere\n. \n\n\nDesign",
"title": "Home"
},
{
"location": "/index.html#what-is-coach",
"text": "",
"title": "What is Coach?"
},
{
"location": "/index.html#motivation",
"text": "Train and evaluate reinforcement learning agents by harnessing the power of multi-core CPU processing to achieve state-of-the-art results. Provide a sandbox for easing the development process of new algorithms through a modular design and an elegant set of APIs.",
"title": "Motivation"
},
{
"location": "/index.html#solution",
"text": "Coach is a python environment which models the interaction between an agent and an environment in a modular way.\nWith Coach, it is possible to model an agent by combining various building blocks, and training the agent on multiple environments.\nThe available environments allow testing the agent in different practical fields such as robotics, autonomous driving, games and more. \nCoach collects statistics from the training process and supports advanced visualization techniques for debugging the agent being trained. Blog post from the Intel\u00ae Nervana\u2122 website can be found here . GitHub repository is here .",
"title": "Solution"
},
{
"location": "/index.html#design",
"text": "",
"title": "Design"
},
{
"location": "/design/index.html",
"text": "Coach Design\n\n\nNetwork Design\n\n\nEach agent has at least one neural network, used as the function approximator, for choosing the actions. The network is designed in a modular way to allow reusability in different agents. It is separated into three main parts:\n\n\n\n\n\n\nInput Embedders\n - This is the first stage of the network, meant to convert the input into a feature vector representation. It is possible to combine several instances of any of the supported embedders, in order to allow varied combinations of inputs. \n\n\nThere are two main types of input embedders: \n\n\n\n\nImage embedder - Convolutional neural network. \n\n\nVector embedder - Multi-layer perceptron. \n\n\n\n\n\n\n\n\nMiddlewares\n - The middleware gets the output of the input embedder, and processes it into a different representation domain, before sending it through the output head. The goal of the middleware is to enable processing the combined outputs of several input embedders, and pass them through some extra processing. This, for instance, might include an LSTM or just a plain simple FC layer.\n\n\n\n\n\n\nOutput Heads\n - The output head is used in order to predict the values required from the network. These might include action-values, state-values or a policy. As with the input embedders, it is possible to use several output heads in the same network. For example, the \nActor Critic\n agent combines two heads - a policy head and a state-value head.\n In addition, the output heads defines the loss function according to the head type.\n\n\n\n\n\n\n\u200b\n\n\n\n\n\n\n\n\n\n\n\nKeeping Network Copies in Sync\n\n\nMost of the reinforcement learning agents include more than one copy of the neural network. These copies serve as counterparts of the main network which are updated in different rates, and are often synchronized either locally or between parallel workers. For easier synchronization of those copies, a wrapper around these copies exposes a simplified API, which allows hiding these complexities from the agent. \n\n\n\n\n\n\n\n\n\n\n\nSupported Algorithms\n\n\nCoach supports many state-of-the-art reinforcement learning algorithms, which are separated into two main classes - value optimization and policy optimization. A detailed description of those algorithms may be found in the algorithms section.",
"title": "Design"
},
{
"location": "/design/index.html#coach-design",
"text": "",
"title": "Coach Design"
},
{
"location": "/design/index.html#network-design",
"text": "Each agent has at least one neural network, used as the function approximator, for choosing the actions. The network is designed in a modular way to allow reusability in different agents. It is separated into three main parts: Input Embedders - This is the first stage of the network, meant to convert the input into a feature vector representation. It is possible to combine several instances of any of the supported embedders, in order to allow varied combinations of inputs. There are two main types of input embedders: Image embedder - Convolutional neural network. Vector embedder - Multi-layer perceptron. Middlewares - The middleware gets the output of the input embedder, and processes it into a different representation domain, before sending it through the output head. The goal of the middleware is to enable processing the combined outputs of several input embedders, and pass them through some extra processing. This, for instance, might include an LSTM or just a plain simple FC layer. Output Heads - The output head is used in order to predict the values required from the network. These might include action-values, state-values or a policy. As with the input embedders, it is possible to use several output heads in the same network. For example, the Actor Critic agent combines two heads - a policy head and a state-value head.\n In addition, the output heads defines the loss function according to the head type. \u200b",
"title": "Network Design"
},
{
"location": "/design/index.html#keeping-network-copies-in-sync",
"text": "Most of the reinforcement learning agents include more than one copy of the neural network. These copies serve as counterparts of the main network which are updated in different rates, and are often synchronized either locally or between parallel workers. For easier synchronization of those copies, a wrapper around these copies exposes a simplified API, which allows hiding these complexities from the agent.",
"title": "Keeping Network Copies in Sync"
},
{
"location": "/design/index.html#supported-algorithms",
"text": "Coach supports many state-of-the-art reinforcement learning algorithms, which are separated into two main classes - value optimization and policy optimization. A detailed description of those algorithms may be found in the algorithms section.",
"title": "Supported Algorithms"
},
{
"location": "/usage/index.html",
"text": "Coach Usage\n\n\nTraining an Agent\n\n\nSingle-threaded Algorithms\n\n\nThis is the most common case. Just choose a preset using the \n-p\n flag and press enter.\n\n\nExample:\n\n\npython coach.py -p CartPole_DQN\n\n\nMulti-threaded Algorithms\n\n\nMulti-threaded algorithms are very common this days.\nThey typically achieve the best results, and scale gracefully with the number of threads.\nIn Coach, running such algorithms is done by selecting a suitable preset, and choosing the number of threads to run using the \n-n\n flag.\n\n\nExample:\n\n\npython coach.py -p CartPole_A3C -n 8\n\n\nEvaluating an Agent\n\n\nThere are several options for evaluating an agent during the training:\n\n\n\n\n\n\nFor multi-threaded runs, an evaluation agent will constantly run in the background and evaluate the model during the training.\n\n\n\n\n\n\nFor single-threaded runs, it is possible to define an evaluation period through the preset. This will run several episodes of evaluation once in a while.\n\n\n\n\n\n\nAdditionally, it is possible to save checkpoints of the agents networks and then run only in evaluation mode.\nSaving checkpoints can be done by specifying the number of seconds between storing checkpoints using the \n-s\n flag.\nThe checkpoints will be saved into the experiment directory.\nLoading a model for evaluation can be done by specifying the \n-crd\n flag with the experiment directory, and the \n--evaluate\n flag to disable training.\n\n\nExample:\n\n\npython coach.py -p CartPole_DQN -s 60\n\n\npython coach.py -p CartPole_DQN --evaluate -crd CHECKPOINT_RESTORE_DIR\n\n\nPlaying with the Environment as a Human\n\n\nInteracting with the environment as a human can be useful for understanding its difficulties and for collecting data for imitation learning.\nIn Coach, this can be easily done by selecting a preset that defines the environment to use, and specifying the \n--play\n flag.\nWhen the environment is loaded, the available keyboard buttons will be printed to the screen.\nPressing the escape key when finished will end the simulation and store the replay buffer in the experiment dir.\n\n\nExample:\n\n\npython coach.py -p Breakout_DQN --play\n\n\nLearning Through Imitation Learning\n\n\nLearning through imitation of human behavior is a nice way to speedup the learning.\nIn Coach, this can be done in two steps -\n\n\n\n\n\n\nCreate a dataset of demonstrations by playing with the environment as a human.\n After this step, a pickle of the replay buffer containing your game play will be stored in the experiment directory.\n The path to this replay buffer will be printed to the screen.\n To do so, you should select an environment type and level through the command line, and specify the \n--play\n flag.\n\n\nExample:\n\n\npython coach.py -et Doom -lvl Basic --play\n\n\n\n\n\n\nNext, use an imitation learning preset and set the replay buffer path accordingly.\n The path can be set either from the command line or from the preset itself.\n\n\nExample:\n\n\npython coach.py -p Doom_Basic_BC -cp='agent.load_memory_from_file_path=\\\"<experiment dir>/replay_buffer.p\\\"'\n\n\n\n\n\n\nVisualizations\n\n\nRendering the Environment\n\n\nRendering the environment can be done by using the \n-r\n flag.\nWhen working with multi-threaded algorithms, the rendered image will be representing the game play of the evaluation worker.\nWhen working with single-threaded algorithms, the rendered image will be representing the single worker which can be either training or evaluating.\nKeep in mind that rendering the environment in single-threaded algorithms may slow the training to some extent.\nWhen playing with the environment using the \n--play\n flag, the environment will be rendered automatically without the need for specifying the \n-r\n flag.\n\n\nExample:\n\n\npython coach.py -p Breakout_DQN -r\n\n\nDumping GIFs\n\n\nCoach allows storing GIFs of the agent game play.\nTo dump GIF files, use the \n-dg\n flag.\nThe files are dumped after every evaluation episode, and are saved into the experiment directory, under a gifs sub-directory.\n\n\nExample:\n\n\npython coach.py -p Breakout_A3C -n 4 -dg\n\n\nSwitching between deep learning frameworks\n\n\nCoach uses TensorFlow as its main backend framework, but it also supports neon for some of the algorithms.\nBy default, TensorFlow will be used. It is possible to switch to neon using the \n-f\n flag.\n\n\nExample:\n\n\npython coach.py -p Doom_Basic_DQN -f neon\n\n\nAdditional Flags\n\n\nThere are several convenient flags which are important to know about.\nHere we will list most of the flags, but these can be updated from time to time.\nThe most up to date description can be found by using the \n-h\n flag.\n\n\n\n\n\n\n\n\nFlag\n\n\nType\n\n\nDescription\n\n\n\n\n\n\n\n\n\n\n-p PRESET\n, \n`--preset PRESET\n\n\nstring\n\n\nName of a preset to run (as configured in presets.py)\n\n\n\n\n\n\n-l\n, \n--list\n\n\nflag\n\n\nList all available presets\n\n\n\n\n\n\n-e ELine truncated
"title": "Usage"
},
{
"location": "/usage/index.html#coach-usage",
"text": "",
"title": "Coach Usage"
},
{
"location": "/usage/index.html#training-an-agent",
"text": "Single-threaded Algorithms This is the most common case. Just choose a preset using the -p flag and press enter. Example: python coach.py -p CartPole_DQN Multi-threaded Algorithms Multi-threaded algorithms are very common this days.\nThey typically achieve the best results, and scale gracefully with the number of threads.\nIn Coach, running such algorithms is done by selecting a suitable preset, and choosing the number of threads to run using the -n flag. Example: python coach.py -p CartPole_A3C -n 8",
"title": "Training an Agent"
},
{
"location": "/usage/index.html#evaluating-an-agent",
"text": "There are several options for evaluating an agent during the training: For multi-threaded runs, an evaluation agent will constantly run in the background and evaluate the model during the training. For single-threaded runs, it is possible to define an evaluation period through the preset. This will run several episodes of evaluation once in a while. Additionally, it is possible to save checkpoints of the agents networks and then run only in evaluation mode.\nSaving checkpoints can be done by specifying the number of seconds between storing checkpoints using the -s flag.\nThe checkpoints will be saved into the experiment directory.\nLoading a model for evaluation can be done by specifying the -crd flag with the experiment directory, and the --evaluate flag to disable training. Example: python coach.py -p CartPole_DQN -s 60 python coach.py -p CartPole_DQN --evaluate -crd CHECKPOINT_RESTORE_DIR",
"title": "Evaluating an Agent"
},
{
"location": "/usage/index.html#playing-with-the-environment-as-a-human",
"text": "Interacting with the environment as a human can be useful for understanding its difficulties and for collecting data for imitation learning.\nIn Coach, this can be easily done by selecting a preset that defines the environment to use, and specifying the --play flag.\nWhen the environment is loaded, the available keyboard buttons will be printed to the screen.\nPressing the escape key when finished will end the simulation and store the replay buffer in the experiment dir. Example: python coach.py -p Breakout_DQN --play",
"title": "Playing with the Environment as a Human"
},
{
"location": "/usage/index.html#learning-through-imitation-learning",
"text": "Learning through imitation of human behavior is a nice way to speedup the learning.\nIn Coach, this can be done in two steps - Create a dataset of demonstrations by playing with the environment as a human.\n After this step, a pickle of the replay buffer containing your game play will be stored in the experiment directory.\n The path to this replay buffer will be printed to the screen.\n To do so, you should select an environment type and level through the command line, and specify the --play flag. Example: python coach.py -et Doom -lvl Basic --play Next, use an imitation learning preset and set the replay buffer path accordingly.\n The path can be set either from the command line or from the preset itself. Example: python coach.py -p Doom_Basic_BC -cp='agent.load_memory_from_file_path=\\\"<experiment dir>/replay_buffer.p\\\"'",
"title": "Learning Through Imitation Learning"
},
{
"location": "/usage/index.html#visualizations",
"text": "Rendering the Environment Rendering the environment can be done by using the -r flag.\nWhen working with multi-threaded algorithms, the rendered image will be representing the game play of the evaluation worker.\nWhen working with single-threaded algorithms, the rendered image will be representing the single worker which can be either training or evaluating.\nKeep in mind that rendering the environment in single-threaded algorithms may slow the training to some extent.\nWhen playing with the environment using the --play flag, the environment will be rendered automatically without the need for specifying the -r flag. Example: python coach.py -p Breakout_DQN -r Dumping GIFs Coach allows storing GIFs of the agent game play.\nTo dump GIF files, use the -dg flag.\nThe files are dumped after every evaluation episode, and are saved into the experiment directory, under a gifs sub-directory. Example: python coach.py -p Breakout_A3C -n 4 -dg",
"title": "Visualizations"
},
{
"location": "/usage/index.html#switching-between-deep-learning-frameworks",
"text": "Coach uses TensorFlow as its main backend framework, but it also supports neon for some of the algorithms.\nBy default, TensorFlow will be used. It is possible to switch to neon using the -f flag. Example: python coach.py -p Doom_Basic_DQN -f neon",
"title": "Switching between deep learning frameworks"
},
{
"location": "/usage/index.html#additional-flags",
"text": "There are several convenient flags which are important to know about.\nHere we will list most of the flags, but these can be updated from time to time.\nThe most up to date description can be found by using the -h flag. Flag Type Description -p PRESET , `--preset PRESET string Name of a preset to run (as configured in presets.py) -l , --list flag List all available presets -e EXPERIMENT_NAME , --experiment_name EXPERIMENT_NAME string Experiment name to be used to store the results. -r , --render flag Render environment -f FRAMEWORK , --framework FRAMEWORK string Neural network framework. Available values: tensorflow, neon -n NUM_WORKERS , --num_workers NUM_WORKERS int Number of workers for multi-process based agents, e.g. A3C --play flag Play as a human by controlling the game with the keyboard. This option will save a replay buffer with the game play. --evaluate flag Run evaluation only. This is a convenient way to disable training in order to evaluate an existing checkpoint. -v , --verbose flag Don't suppress TensorFlow debug prints. -s SAVE_MODEL_SEC , --save_model_sec SAVE_MODEL_SEC int Time in seconds between saving checkpoints of the model. -crd CHECKPOINT_RESTORE_DIR , --checkpoint_restore_dir CHECKPOINT_RESTORE_DIR string Path to a folder containing a checkpoint to restore the model from. -dg , --dump_gifs flag Enable the gif saving functionality. -at AGENT_TYPE , --agent_type AGENT_TYPE string Choose an agent type class to override on top of the selected preset. If no preset is defined, a preset can be set from the command-line by combining settings which are set by using --agent_type , --experiment_type , --environemnt_type -et ENVIRONMENT_TYPE , --environment_type ENVIRONMENT_TYPE string Choose an environment type class to override on top of the selected preset. If no preset is defined, a preset can be set from the command-line by combining settings which are set by using --agent_type , --experiment_type , --environemnt_type -ept EXPLORATION_POLICY_TYPE , --exploration_policy_type EXPLORATION_POLICY_TYPE string Choose an exploration policy type class to override on top of the selected preset.If no preset is defined, a preset can be set from the command-line by combining settings which are set by using --agent_type , --experiment_type , --environemnt_type -lvl LEVEL , --level LEVEL string Choose the level that will be played in the environment that was selected. This value will override the level parameter in the environment class. -cp CUSTOM_PARAMETER , --custom_parameter CUSTOM_PARAMETER string Semicolon separated parameters used to override specific parameters on top of the selected preset (or on top of the command-line assembled one). Whenever a parameter value is a string, it should be inputted as '\\\"string\\\"' . For ex.: \"visualization.render=False; num_training_iterations=500; optimizer='rmsprop'\"",
"title": "Additional Flags"
},
{
"location": "/algorithms/value_optimization/dqn/index.html",
"text": "Deep Q Networks\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nPlaying Atari with Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\nUsing the next states from the sampled batch, run the target network to calculate the \n Q \n values for each of the actions \n Q(s_{t+1},a) \n, and keep only the maximum value for each state. \n\n\nIn order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. \n\n\n\n\nFor each action that was played, use the following equation for calculating the targets of the network:\u200b \n y_t=r(s_t,a_t)+\u03b3\\cdot max_a {Q(s_{t+1},a)} \n\n\n\n\n\n\n\n\nFinally, train the online network using the current states as inputs, and with the aforementioned targets. \n\n\n\n\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "DQN"
},
{
"location": "/algorithms/value_optimization/dqn/index.html#deep-q-networks",
"text": "Actions space: Discrete References: Playing Atari with Deep Reinforcement Learning",
"title": "Deep Q Networks"
},
{
"location": "/algorithms/value_optimization/dqn/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/dqn/index.html#algorithm-description",
"text": "Training the network Sample a batch of transitions from the replay buffer. Using the next states from the sampled batch, run the target network to calculate the Q values for each of the actions Q(s_{t+1},a) , and keep only the maximum value for each state. In order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. For each action that was played, use the following equation for calculating the targets of the network:\u200b y_t=r(s_t,a_t)+\u03b3\\cdot max_a {Q(s_{t+1},a)} Finally, train the online network using the current states as inputs, and with the aforementioned targets. Once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/double_dqn/index.html",
"text": "Double DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nDeep Reinforcement Learning with Double Q-learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\nUsing the next states from the sampled batch, run the online network in order to find the \nQ\n maximizing action \nargmax_a Q(s_{t+1},a)\n. For these actions, use the corresponding next states and run the target network to calculate \nQ(s_{t+1},argmax_a Q(s_{t+1},a))\n.\n\n\nIn order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. \n\n\n\n\nFor each action that was played, use the following equation for calculating the targets of the network:\n \n y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},argmax_a Q(s_{t+1},a)) \n\n\n\n\n\n\n\n\nFinally, train the online network using the current states as inputs, and with the aforementioned targets. \n\n\n\n\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Double DQN"
},
{
"location": "/algorithms/value_optimization/double_dqn/index.html#double-dqn",
"text": "Actions space: Discrete References: Deep Reinforcement Learning with Double Q-learning",
"title": "Double DQN"
},
{
"location": "/algorithms/value_optimization/double_dqn/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/double_dqn/index.html#algorithm-description",
"text": "Training the network Sample a batch of transitions from the replay buffer. Using the next states from the sampled batch, run the online network in order to find the Q maximizing action argmax_a Q(s_{t+1},a) . For these actions, use the corresponding next states and run the target network to calculate Q(s_{t+1},argmax_a Q(s_{t+1},a)) . In order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. For each action that was played, use the following equation for calculating the targets of the network:\n y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},argmax_a Q(s_{t+1},a)) Finally, train the online network using the current states as inputs, and with the aforementioned targets. Once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/index.html",
"text": "Dueling DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nDueling Network Architectures for Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nGeneral Description\n\n\nDueling DQN presents a change in the network structure comparing to DQN.\n\n\nDueling DQN uses a specialized \nDueling Q Head\n in order to separate \n Q \n to an \n A \n (advantage) stream and a \n V \n stream. Adding this type of structure to the network head allows the network to better differentiate actions from one another, and significantly improves the learning.\n\n\nIn many states, the values of the different actions are very similar, and it is less important which action to take.\nThis is especially important in environments where there are many actions to choose from. In DQN, on each training iteration, for each of the states in the batch, we update the \nQ\n values only for the specific actions taken in those states. This results in slower learning as we do not learn the \nQ\n values for actions that were not taken yet. On dueling architecture, on the other hand, learning is faster - as we start learning the state-value even if only a single action has been taken at this state.",
"title": "Dueling DQN"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/index.html#dueling-dqn",
"text": "Actions space: Discrete References: Dueling Network Architectures for Deep Reinforcement Learning",
"title": "Dueling DQN"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/index.html#general-description",
"text": "Dueling DQN presents a change in the network structure comparing to DQN. Dueling DQN uses a specialized Dueling Q Head in order to separate Q to an A (advantage) stream and a V stream. Adding this type of structure to the network head allows the network to better differentiate actions from one another, and significantly improves the learning. In many states, the values of the different actions are very similar, and it is less important which action to take.\nThis is especially important in environments where there are many actions to choose from. In DQN, on each training iteration, for each of the states in the batch, we update the Q values only for the specific actions taken in those states. This results in slower learning as we do not learn the Q values for actions that were not taken yet. On dueling architecture, on the other hand, learning is faster - as we start learning the state-value even if only a single action has been taken at this state.",
"title": "General Description"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/index.html",
"text": "Categorical DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nA Distributional Perspective on Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\n\n\nThe Bellman update is projected to the set of atoms representing the \n Q \n values distribution, such that the \ni-th\n component of the projected update is calculated as follows:\n \n (\\Phi \\hat{T} Z_{\\theta}(s_t,a_t))_i=\\sum_{j=0}^{N-1}\\Big[1-\\frac{|[\\hat{T}_{z_{j}}]^{V_{MAX}}_{V_{MIN}}-z_i|}{\\Delta z}\\Big]^1_0 \\ p_j(s_{t+1}, \\pi(s_{t+1})) \n\n where:\n\n\n\n\n\n\n[ \\cdot ] \n bounds its argument in the range [a, b]\n\n\n\n\n\\hat{T}_{z_{j}}\n is the Bellman update for atom \nz_j\n: \u00a0 \u00a0 \n\\hat{T}_{z_{j}} := r+\\gamma z_j\n\n\n\n\n\n\n\n\n\n\nNetwork is trained with the cross entropy loss between the resulting probability distribution and the target probability distribution. Only the target of the actions that were actually taken is updated. \n\n\n\n\nOnce in every few thousand steps, weights are copied from the online network to the target network.",
"title": "Categorical DQN"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/index.html#categorical-dqn",
"text": "Actions space: Discrete References: A Distributional Perspective on Reinforcement Learning",
"title": "Categorical DQN"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/index.html#algorithm-description",
"text": "Training the network Sample a batch of transitions from the replay buffer. The Bellman update is projected to the set of atoms representing the Q values distribution, such that the i-th component of the projected update is calculated as follows:\n (\\Phi \\hat{T} Z_{\\theta}(s_t,a_t))_i=\\sum_{j=0}^{N-1}\\Big[1-\\frac{|[\\hat{T}_{z_{j}}]^{V_{MAX}}_{V_{MIN}}-z_i|}{\\Delta z}\\Big]^1_0 \\ p_j(s_{t+1}, \\pi(s_{t+1})) \n where: [ \\cdot ] bounds its argument in the range [a, b] \\hat{T}_{z_{j}} is the Bellman update for atom z_j : \u00a0 \u00a0 \\hat{T}_{z_{j}} := r+\\gamma z_j Network is trained with the cross entropy loss between the resulting probability distribution and the target probability distribution. Only the target of the actions that were actually taken is updated. Once in every few thousand steps, weights are copied from the online network to the target network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/mmc/index.html",
"text": "Mixed Monte Carlo\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nCount-Based Exploration with Neural Density Models\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\nIn MMC, targets are calculated as a mixture between Double DQN targets and full Monte Carlo samples (total discounted returns).\n\n\nThe DDQN targets are calculated in the same manner as in the DDQN agent:\n\n\n\n\n y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) \n\n\n\n\nThe Monte Carlo targets are calculated by summing up the discounted rewards across the entire episode:\n\n\n\n\n y_t^{MC}=\\sum_{j=0}^T\\gamma^j r(s_{t+j},a_{t+j} ) \n\n\n\n\nA mixing ratio \n\\alpha\n is then used to get the final targets:\n\n\n\n\n y_t=(1-\\alpha)\\cdot y_t^{DDQN}+\\alpha \\cdot y_t^{MC} \n\n\n\n\nFinally, the online network is trained using the current states as inputs, and the calculated targets.\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Mixed Monte Carlo"
},
{
"location": "/algorithms/value_optimization/mmc/index.html#mixed-monte-carlo",
"text": "Actions space: Discrete References: Count-Based Exploration with Neural Density Models",
"title": "Mixed Monte Carlo"
},
{
"location": "/algorithms/value_optimization/mmc/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/mmc/index.html#algorithm-description",
"text": "Training the network In MMC, targets are calculated as a mixture between Double DQN targets and full Monte Carlo samples (total discounted returns). The DDQN targets are calculated in the same manner as in the DDQN agent: y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) The Monte Carlo targets are calculated by summing up the discounted rewards across the entire episode: y_t^{MC}=\\sum_{j=0}^T\\gamma^j r(s_{t+j},a_{t+j} ) A mixing ratio \\alpha is then used to get the final targets: y_t=(1-\\alpha)\\cdot y_t^{DDQN}+\\alpha \\cdot y_t^{MC} Finally, the online network is trained using the current states as inputs, and the calculated targets.\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/pal/index.html",
"text": "Persistent Advantage Learning\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nIncreasing the Action Gap: New Operators for Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\n\n\n\n\nStart by calculating the initial target values in the same manner as they are calculated in DDQN\n \n y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) \n\n\n\n\n\n\nThe action gap \n V(s_t )-Q(s_t,a_t) \n should then be subtracted from each of the calculated targets. To calculate the action gap, run the target network using the current states and get the \n Q \n values for all the actions. Then estimate \n V \n as the maximum predicted \n Q \n value for the current state:\n \n V(s_t )=max_a Q(s_t,a) \n\n\n\n\nFor \nadvantage learning (AL)\n, reduce the action gap weighted by a predefined parameter \n \\alpha \n from the targets \n y_t^{DDQN} \n: \n \n y_t=y_t^{DDQN}-\\alpha \\cdot (V(s_t )-Q(s_t,a_t )) \n\n\n\n\nFor \npersistent advantage learning (PAL)\n, the target network is also used in order to calculate the action gap for the next state:\n \n V(s_{t+1} )-Q(s_{t+1},a_{t+1}) \n\n where \n a_{t+1} \n is chosen by running the next states through the online network and choosing the action that has the highest predicted \n Q \n value. Finally, the targets will be defined as -\n \n y_t=y_t^{DDQN}-\\alpha \\cdot min(V(s_t )-Q(s_t,a_t ),V(s_{t+1} )-Q(s_{t+1},a_{t+1} )) \n\n\n\n\n\n\nTrain the online network using the current states as inputs, and with the aforementioned targets.\n\n\n\n\n\n\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Persistent Advantage Learning"
},
{
"location": "/algorithms/value_optimization/pal/index.html#persistent-advantage-learning",
"text": "Actions space: Discrete References: Increasing the Action Gap: New Operators for Reinforcement Learning",
"title": "Persistent Advantage Learning"
},
{
"location": "/algorithms/value_optimization/pal/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/pal/index.html#algorithm-description",
"text": "Training the network Sample a batch of transitions from the replay buffer. Start by calculating the initial target values in the same manner as they are calculated in DDQN\n y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) The action gap V(s_t )-Q(s_t,a_t) should then be subtracted from each of the calculated targets. To calculate the action gap, run the target network using the current states and get the Q values for all the actions. Then estimate V as the maximum predicted Q value for the current state:\n V(s_t )=max_a Q(s_t,a) For advantage learning (AL) , reduce the action gap weighted by a predefined parameter \\alpha from the targets y_t^{DDQN} : \n y_t=y_t^{DDQN}-\\alpha \\cdot (V(s_t )-Q(s_t,a_t )) For persistent advantage learning (PAL) , the target network is also used in order to calculate the action gap for the next state:\n V(s_{t+1} )-Q(s_{t+1},a_{t+1}) \n where a_{t+1} is chosen by running the next states through the online network and choosing the action that has the highest predicted Q value. Finally, the targets will be defined as -\n y_t=y_t^{DDQN}-\\alpha \\cdot min(V(s_t )-Q(s_t,a_t ),V(s_{t+1} )-Q(s_{t+1},a_{t+1} )) Train the online network using the current states as inputs, and with the aforementioned targets. Once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/nec/index.html",
"text": "Neural Episodic Control\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nNeural Episodic Control\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\n\n\nUse the current state as an input to the online network and extract the state embedding, which is the intermediate output from the middleware. \n\n\nFor each possible action \na_i\n, run the DND head using the state embedding and the selected action \na_i\n as inputs. The DND is queried and returns the \n P \n nearest neighbor keys and values. The keys and values are used to calculate and return the action \n Q \n value from the network. \n\n\nPass all the \n Q \n values to the exploration policy and choose an action accordingly. \n\n\nStore the state embeddings and actions taken during the current episode in a small buffer \nB\n, in order to accumulate transitions until it is possible to calculate the total discounted returns over the entire episode.\n\n\n\n\nFinalizing an episode\n\n\nFor each step in the episode, the state embeddings and the taken actions are stored in the buffer \nB\n. When the episode is finished, the replay buffer calculates the \n N \n-step total return of each transition in the buffer, bootstrapped using the maximum \nQ\n value of the \nN\n-th transition. Those values are inserted along with the total return into the DND, and the buffer \nB\n is reset.\n\n\nTraining the network\n\n\nTrain the network only when the DND has enough entries for querying.\n\n\nTo train the network, the current states are used as the inputs and the \nN\n-step returns are used as the targets. The \nN\n-step return used takes into account \n N \n consecutive steps, and bootstraps the last value from the network if necessary:\n\n y_t=\\sum_{j=0}^{N-1}\\gamma^j r(s_{t+j},a_{t+j} ) +\\gamma^N max_a Q(s_{t+N},a)",
"title": "Neural Episodic Control"
},
{
"location": "/algorithms/value_optimization/nec/index.html#neural-episodic-control",
"text": "Actions space: Discrete References: Neural Episodic Control",
"title": "Neural Episodic Control"
},
{
"location": "/algorithms/value_optimization/nec/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/nec/index.html#algorithm-description",
"text": "Choosing an action Use the current state as an input to the online network and extract the state embedding, which is the intermediate output from the middleware. For each possible action a_i , run the DND head using the state embedding and the selected action a_i as inputs. The DND is queried and returns the P nearest neighbor keys and values. The keys and values are used to calculate and return the action Q value from the network. Pass all the Q values to the exploration policy and choose an action accordingly. Store the state embeddings and actions taken during the current episode in a small buffer B , in order to accumulate transitions until it is possible to calculate the total discounted returns over the entire episode. Finalizing an episode For each step in the episode, the state embeddings and the taken actions are stored in the buffer B . When the episode is finished, the replay buffer calculates the N -step total return of each transition in the buffer, bootstrapped using the maximum Q value of the N -th transition. Those values are inserted along with the total return into the DND, and the buffer B is reset. Training the network Train the network only when the DND has enough entries for querying. To train the network, the current states are used as the inputs and the N -step returns are used as the targets. The N -step return used takes into account N consecutive steps, and bootstraps the last value from the network if necessary: y_t=\\sum_{j=0}^{N-1}\\gamma^j r(s_{t+j},a_{t+j} ) +\\gamma^N max_a Q(s_{t+N},a)",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/bs_dqn/index.html",
"text": "Bootstrapped DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nDeep Exploration via Bootstrapped DQN\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\nThe current states are used as the input to the network. The network contains several \nQ\n heads, which are used for returning different estimations of the action \n Q \n values. For each episode, the bootstrapped exploration policy selects a single head to play with during the episode. According to the selected head, only the relevant output \n Q \n values are used. Using those \n Q \n values, the exploration policy then selects the action for acting.\n\n\nStoring the transitions\n\n\nFor each transition, a Binomial mask is generated according to a predefined probability, and the number of output heads. The mask is a binary vector where each element holds a 0 for heads that shouldn't train on the specific transition, and 1 for heads that should use the transition for training. The mask is stored as part of the transition info in the replay buffer. \n\n\nTraining the network\n\n\nFirst, sample a batch of transitions from the replay buffer. Run the current states through the network and get the current \n Q \n value predictions for all the heads and all the actions. For each transition in the batch, and for each output head, if the transition mask is 1 - change the targets of the played action to \ny_t\n, according to the standard DQN update rule:\n\n\n\n\n y_t=r(s_t,a_t )+\\gamma\\cdot max_a Q(s_{t+1},a) \n\n\n\n\nOtherwise, leave it intact so that the transition does not affect the learning of this head. Then, train the online network according to the calculated targets.\n\n\nAs in DQN, once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Bootstrapped DQN"
},
{
"location": "/algorithms/value_optimization/bs_dqn/index.html#bootstrapped-dqn",
"text": "Actions space: Discrete References: Deep Exploration via Bootstrapped DQN",
"title": "Bootstrapped DQN"
},
{
"location": "/algorithms/value_optimization/bs_dqn/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/bs_dqn/index.html#algorithm-description",
"text": "Choosing an action The current states are used as the input to the network. The network contains several Q heads, which are used for returning different estimations of the action Q values. For each episode, the bootstrapped exploration policy selects a single head to play with during the episode. According to the selected head, only the relevant output Q values are used. Using those Q values, the exploration policy then selects the action for acting. Storing the transitions For each transition, a Binomial mask is generated according to a predefined probability, and the number of output heads. The mask is a binary vector where each element holds a 0 for heads that shouldn't train on the specific transition, and 1 for heads that should use the transition for training. The mask is stored as part of the transition info in the replay buffer. Training the network First, sample a batch of transitions from the replay buffer. Run the current states through the network and get the current Q value predictions for all the heads and all the actions. For each transition in the batch, and for each output head, if the transition mask is 1 - change the targets of the played action to y_t , according to the standard DQN update rule: y_t=r(s_t,a_t )+\\gamma\\cdot max_a Q(s_{t+1},a) Otherwise, leave it intact so that the transition does not affect the learning of this head. Then, train the online network according to the calculated targets. As in DQN, once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/n_step/index.html",
"text": "N-Step Q Learning\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nAsynchronous Methods for Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\nThe \nN\n-step Q learning algorithm works in similar manner to DQN except for the following changes:\n\n\n\n\n\n\nNo replay buffer is used. Instead of sampling random batches of transitions, the network is trained every \nN\n steps using the latest \nN\n steps played by the agent.\n\n\n\n\n\n\nIn order to stabilize the learning, multiple workers work together to update the network. This creates the same effect as uncorrelating the samples used for training.\n\n\n\n\n\n\nInstead of using single-step Q targets for the network, the rewards from \nN\n consequent steps are accumulated to form the \nN\n-step Q targets, according to the following equation: \n\nR(s_t, a_t) = \\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k})\n\nwhere \nk\n is \nT_{max} - State\\_Index\n for each state in the batch",
"title": "N-Step Q Learning"
},
{
"location": "/algorithms/value_optimization/n_step/index.html#n-step-q-learning",
"text": "Actions space: Discrete References: Asynchronous Methods for Deep Reinforcement Learning",
"title": "N-Step Q Learning"
},
{
"location": "/algorithms/value_optimization/n_step/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/n_step/index.html#algorithm-description",
"text": "Training the network The N -step Q learning algorithm works in similar manner to DQN except for the following changes: No replay buffer is used. Instead of sampling random batches of transitions, the network is trained every N steps using the latest N steps played by the agent. In order to stabilize the learning, multiple workers work together to update the network. This creates the same effect as uncorrelating the samples used for training. Instead of using single-step Q targets for the network, the rewards from N consequent steps are accumulated to form the N -step Q targets, according to the following equation: R(s_t, a_t) = \\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k}) \nwhere k is T_{max} - State\\_Index for each state in the batch",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/naf/index.html",
"text": "Normalized Advantage Functions\n\n\nActions space:\n Continuous\n\n\nReferences:\n \nContinuous Deep Q-Learning with Model-based Acceleration\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\nThe current state is used as an input to the network. The action mean \n \\mu(s_t ) \n is extracted from the output head. It is then passed to the exploration policy which adds noise in order to encourage exploration.\n\n\nTraining the network\n\n\nThe network is trained by using the following targets:\n\n y_t=r(s_t,a_t )+\\gamma\\cdot V(s_{t+1}) \n\nUse the next states as the inputs to the target network and extract the \n V \n value, from within the head, to get \n V(s_{t+1} ) \n. Then, update the online network using the current states and actions as inputs, and \n y_t \n as the targets.\nAfter every training step, use a soft update in order to copy the weights from the online network to the target network.",
"title": "Normalized Advantage Functions"
},
{
"location": "/algorithms/value_optimization/naf/index.html#normalized-advantage-functions",
"text": "Actions space: Continuous References: Continuous Deep Q-Learning with Model-based Acceleration",
"title": "Normalized Advantage Functions"
},
{
"location": "/algorithms/value_optimization/naf/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/naf/index.html#algorithm-description",
"text": "Choosing an action The current state is used as an input to the network. The action mean \\mu(s_t ) is extracted from the output head. It is then passed to the exploration policy which adds noise in order to encourage exploration. Training the network The network is trained by using the following targets: y_t=r(s_t,a_t )+\\gamma\\cdot V(s_{t+1}) \nUse the next states as the inputs to the target network and extract the V value, from within the head, to get V(s_{t+1} ) . Then, update the online network using the current states and actions as inputs, and y_t as the targets.\nAfter every training step, use a soft update in order to copy the weights from the online network to the target network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/pg/index.html",
"text": "Policy Gradient\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nSimple Statistical Gradient-Following Algorithms for Connectionist Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Discrete actions\n\n\nRun the current states through the network and get a policy distribution over the actions. While training, sample from the policy distribution. When testing, take the action with the highest probability. \n\n\nTraining the network\n\n\nThe policy head loss is defined as \n L=-log (\\pi) \\cdot PolicyGradientRescaler \n. The \nPolicyGradientRescaler\n is used in order to reduce the policy gradient variance, which might be very noisy. This is done in order to reduce the variance of the updates, since noisy gradient updates might destabilize the policy's convergence. The rescaler is a configurable parameter and there are few options to choose from: \n\n\n \nTotal Episode Return\n - The sum of all the discounted rewards during the episode.\n\n \nFuture Return\n - Return from each transition until the end of the episode.\n\n \nFuture Return Normalized by Episode\n - Future returns across the episode normalized by the episode's mean and standard deviation.\n\n \nFuture Return Normalized by Timestep\n - Future returns normalized using running means and standard deviations, which are calculated seperately for each timestep, across different episodes. \n\n\nGradients are accumulated over a number of full played episodes. The gradients accumulation over several episodes serves the same purpose - reducing the update variance. After accumulating gradients for several episodes, the gradients are then applied to the network.",
"title": "Policy Gradient"
},
{
"location": "/algorithms/policy_optimization/pg/index.html#policy-gradient",
"text": "Actions space: Discrete|Continuous References: Simple Statistical Gradient-Following Algorithms for Connectionist Reinforcement Learning",
"title": "Policy Gradient"
},
{
"location": "/algorithms/policy_optimization/pg/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/pg/index.html#algorithm-description",
"text": "Choosing an action - Discrete actions Run the current states through the network and get a policy distribution over the actions. While training, sample from the policy distribution. When testing, take the action with the highest probability. Training the network The policy head loss is defined as L=-log (\\pi) \\cdot PolicyGradientRescaler . The PolicyGradientRescaler is used in order to reduce the policy gradient variance, which might be very noisy. This is done in order to reduce the variance of the updates, since noisy gradient updates might destabilize the policy's convergence. The rescaler is a configurable parameter and there are few options to choose from: Total Episode Return - The sum of all the discounted rewards during the episode. Future Return - Return from each transition until the end of the episode. Future Return Normalized by Episode - Future returns across the episode normalized by the episode's mean and standard deviation. Future Return Normalized by Timestep - Future returns normalized using running means and standard deviations, which are calculated seperately for each timestep, across different episodes. Gradients are accumulated over a number of full played episodes. The gradients accumulation over several episodes serves the same purpose - reducing the update variance. After accumulating gradients for several episodes, the gradients are then applied to the network.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/ac/index.html",
"text": "Actor-Critic\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nAsynchronous Methods for Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Discrete actions\n\n\nThe policy network is used in order to predict action probabilites. While training, a sample is taken from a categorical distribution assigned with these probabilities. When testing, the action with the highest probability is used.\n\n\nTraining the network\n\n\nA batch of \n T_{max} \n transitions is used, and the advantages are calculated upon it.\n\n\nAdvantages can be calculated by either of the following methods (configured by the selected preset) -\n\n\n\n\nA_VALUE\n - Estimating advantage directly:\n A(s_t, a_t) = \\underbrace{\\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k})}_{Q(s_t, a_t)} - V(s_t) \nwhere \nk\n is \nT_{max} - State\\_Index\n for each state in the batch.\n\n\nGAE\n - By following the \nGeneralized Advantage Estimation\n paper. \n\n\n\n\nThe advantages are then used in order to accumulate gradients according to \n\n L = -\\mathop{\\mathbb{E}} [log (\\pi) \\cdot A]",
"title": "Actor-Critic"
},
{
"location": "/algorithms/policy_optimization/ac/index.html#actor-critic",
"text": "Actions space: Discrete|Continuous References: Asynchronous Methods for Deep Reinforcement Learning",
"title": "Actor-Critic"
},
{
"location": "/algorithms/policy_optimization/ac/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/ac/index.html#algorithm-description",
"text": "Choosing an action - Discrete actions The policy network is used in order to predict action probabilites. While training, a sample is taken from a categorical distribution assigned with these probabilities. When testing, the action with the highest probability is used. Training the network A batch of T_{max} transitions is used, and the advantages are calculated upon it. Advantages can be calculated by either of the following methods (configured by the selected preset) - A_VALUE - Estimating advantage directly: A(s_t, a_t) = \\underbrace{\\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k})}_{Q(s_t, a_t)} - V(s_t) where k is T_{max} - State\\_Index for each state in the batch. GAE - By following the Generalized Advantage Estimation paper. The advantages are then used in order to accumulate gradients according to L = -\\mathop{\\mathbb{E}} [log (\\pi) \\cdot A]",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/ddpg/index.html",
"text": "Deep Deterministic Policy Gradient\n\n\nActions space:\n Continuous\n\n\nReferences:\n \nContinuous control with deep reinforcement learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\nPass the current states through the actor network, and get an action mean vector \n \\mu \n. While in training phase, use a continuous exploration policy, such as the Ornstein-Uhlenbeck process, to add exploration noise to the action. When testing, use the mean vector \n\\mu\n as-is.\n\n\nTraining the network\n\n\nStart by sampling a batch of transitions from the experience replay.\n\n\n\n\nTo train the \ncritic network\n, use the following targets:\n\n\n\n\n\n\n y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},\\mu(s_{t+1} )) \n\n First run the actor target network, using the next states as the inputs, and get \n \\mu (s_{t+1} ) \n. Next, run the critic target network using the next states and \n \\mu (s_{t+1} ) \n, and use the output to calculate \n y_t \n according to the equation above. To train the network, use the current states and actions as the inputs, and \ny_t\n as the targets.\n\n\n\n\nTo train the \nactor network\n, use the following equation:\n\n\n\n\n\n\n \\nabla_{\\theta^\\mu } J \\approx E_{s_t \\tilde{} \\rho^\\beta } [\\nabla_a Q(s,a)|_{s=s_t,a=\\mu (s_t ) } \\cdot \\nabla_{\\theta^\\mu} \\mu(s)|_{s=s_t} ] \n\n Use the actor's online network to get the action mean values using the current states as the inputs. Then, use the critic online network in order to get the gradients of the critic output with respect to the action mean values \n \\nabla _a Q(s,a)|_{s=s_t,a=\\mu(s_t ) } \n. Using the chain rule, calculate the gradients of the actor's output, with respect to the actor weights, given \n \\nabla_a Q(s,a) \n. Finally, apply those gradients to the actor network.\n\n\nAfter every training step, do a soft update of the critic and actor target networks' weights from the online networks.",
"title": "Deep Determinstic Policy Gradients"
},
{
"location": "/algorithms/policy_optimization/ddpg/index.html#deep-deterministic-policy-gradient",
"text": "Actions space: Continuous References: Continuous control with deep reinforcement learning",
"title": "Deep Deterministic Policy Gradient"
},
{
"location": "/algorithms/policy_optimization/ddpg/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/ddpg/index.html#algorithm-description",
"text": "Choosing an action Pass the current states through the actor network, and get an action mean vector \\mu . While in training phase, use a continuous exploration policy, such as the Ornstein-Uhlenbeck process, to add exploration noise to the action. When testing, use the mean vector \\mu as-is. Training the network Start by sampling a batch of transitions from the experience replay. To train the critic network , use the following targets: y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},\\mu(s_{t+1} )) \n First run the actor target network, using the next states as the inputs, and get \\mu (s_{t+1} ) . Next, run the critic target network using the next states and \\mu (s_{t+1} ) , and use the output to calculate y_t according to the equation above. To train the network, use the current states and actions as the inputs, and y_t as the targets. To train the actor network , use the following equation: \\nabla_{\\theta^\\mu } J \\approx E_{s_t \\tilde{} \\rho^\\beta } [\\nabla_a Q(s,a)|_{s=s_t,a=\\mu (s_t ) } \\cdot \\nabla_{\\theta^\\mu} \\mu(s)|_{s=s_t} ] \n Use the actor's online network to get the action mean values using the current states as the inputs. Then, use the critic online network in order to get the gradients of the critic output with respect to the action mean values \\nabla _a Q(s,a)|_{s=s_t,a=\\mu(s_t ) } . Using the chain rule, calculate the gradients of the actor's output, with respect to the actor weights, given \\nabla_a Q(s,a) . Finally, apply those gradients to the actor network. After every training step, do a soft update of the critic and actor target networks' weights from the online networks.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/ppo/index.html",
"text": "Proximal Policy Optimization\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nProximal Policy Optimization Algorithms\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Continuous actions\n\n\nRun the observation through the policy network, and get the mean and standard deviation vectors for this observation. While in training phase, sample from a multi-dimensional Gaussian distribution with these mean and standard deviation values. When testing, just take the mean values predicted by the network. \n\n\nTraining the network\n\n\n\n\nCollect a big chunk of experience (in the order of thousands of transitions, sampled from multiple episodes).\n\n\nCalculate the advantages for each transition, using the \nGeneralized Advantage Estimation\n method (Schulman '2015). \n\n\nRun a single training iteration of the value network using an L-BFGS optimizer. Unlike first order optimizers, the L-BFGS optimizer runs on the entire dataset at once, without batching. It continues running until some low loss threshold is reached. To prevent overfitting to the current dataset, the value targets are updated in a soft manner, using an Exponentially Weighted Moving Average, based on the total discounted returns of each state in each episode.\n\n\nRun several training iterations of the policy network. This is done by using the previously calculated advantages as targets. The loss function penalizes policies that deviate too far from the old policy (the policy that was used \nbefore\n starting to run the current set of training iterations) using a regularization term. \n\n\nAfter training is done, the last sampled KL divergence value will be compared with the \ntarget KL divergence\n value, in order to adapt the penalty coefficient used in the policy loss. If the KL divergence went too high, increase the penalty, if it went too low, reduce it. Otherwise, leave it unchanged.",
"title": "Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/ppo/index.html#proximal-policy-optimization",
"text": "Actions space: Discrete|Continuous References: Proximal Policy Optimization Algorithms",
"title": "Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/ppo/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/ppo/index.html#algorithm-description",
"text": "Choosing an action - Continuous actions Run the observation through the policy network, and get the mean and standard deviation vectors for this observation. While in training phase, sample from a multi-dimensional Gaussian distribution with these mean and standard deviation values. When testing, just take the mean values predicted by the network. Training the network Collect a big chunk of experience (in the order of thousands of transitions, sampled from multiple episodes). Calculate the advantages for each transition, using the Generalized Advantage Estimation method (Schulman '2015). Run a single training iteration of the value network using an L-BFGS optimizer. Unlike first order optimizers, the L-BFGS optimizer runs on the entire dataset at once, without batching. It continues running until some low loss threshold is reached. To prevent overfitting to the current dataset, the value targets are updated in a soft manner, using an Exponentially Weighted Moving Average, based on the total discounted returns of each state in each episode. Run several training iterations of the policy network. This is done by using the previously calculated advantages as targets. The loss function penalizes policies that deviate too far from the old policy (the policy that was used before starting to run the current set of training iterations) using a regularization term. After training is done, the last sampled KL divergence value will be compared with the target KL divergence value, in order to adapt the penalty coefficient used in the policy loss. If the KL divergence went too high, increase the penalty, if it went too low, reduce it. Otherwise, leave it unchanged.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/cppo/index.html",
"text": "Clipped Proximal Policy Optimization\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nProximal Policy Optimization Algorithms\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Continuous action\n\n\nSame as in PPO. \n\n\nTraining the network\n\n\nVery similar to PPO, with several small (but very simplifying) changes:\n\n\n\n\n\n\nTrain both the value and policy networks, simultaneously, by defining a single loss function, which is the sum of each of the networks loss functions. Then, back propagate gradients only once from this unified loss function.\n\n\n\n\n\n\nThe unified network's optimizer is set to Adam (instead of L-BFGS for the value network as in PPO). \n\n\n\n\n\n\nValue targets are now also calculated based on the GAE advantages. In this method, the \n V \n values are predicted from the critic network, and then added to the GAE based advantages, in order to get a \n Q \n value for each action. Now, since our critic network is predicting a \n V \n value for each state, setting the \n Q \n calculated action-values as a target, will on average serve as a \n V \n state-value target. \n\n\n\n\n\n\nInstead of adapting the penalizing KL divergence coefficient used in PPO, the likelihood ratio \nr_t(\\theta) =\\frac{\\pi_{\\theta}(a|s)}{\\pi_{\\theta_{old}}(a|s)}\n is clipped, to achieve a similar effect. This is done by defining the policy's loss function to be the minimum between the standard surrogate loss and an epsilon clipped surrogate loss:\n\n\n\n\n\n\n\n\nL^{CLIP}(\\theta)=E_{t}[min(r_t(\\theta)\\cdot \\hat{A}_t, clip(r_t(\\theta), 1-\\epsilon, 1+\\epsilon) \\cdot \\hat{A}_t)]",
"title": "Clipped Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/cppo/index.html#clipped-proximal-policy-optimization",
"text": "Actions space: Discrete|Continuous References: Proximal Policy Optimization Algorithms",
"title": "Clipped Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/cppo/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/cppo/index.html#algorithm-description",
"text": "Choosing an action - Continuous action Same as in PPO. Training the network Very similar to PPO, with several small (but very simplifying) changes: Train both the value and policy networks, simultaneously, by defining a single loss function, which is the sum of each of the networks loss functions. Then, back propagate gradients only once from this unified loss function. The unified network's optimizer is set to Adam (instead of L-BFGS for the value network as in PPO). Value targets are now also calculated based on the GAE advantages. In this method, the V values are predicted from the critic network, and then added to the GAE based advantages, in order to get a Q value for each action. Now, since our critic network is predicting a V value for each state, setting the Q calculated action-values as a target, will on average serve as a V state-value target. Instead of adapting the penalizing KL divergence coefficient used in PPO, the likelihood ratio r_t(\\theta) =\\frac{\\pi_{\\theta}(a|s)}{\\pi_{\\theta_{old}}(a|s)} is clipped, to achieve a similar effect. This is done by defining the policy's loss function to be the minimum between the standard surrogate loss and an epsilon clipped surrogate loss: L^{CLIP}(\\theta)=E_{t}[min(r_t(\\theta)\\cdot \\hat{A}_t, clip(r_t(\\theta), 1-\\epsilon, 1+\\epsilon) \\cdot \\hat{A}_t)]",
"title": "Algorithm Description"
},
{
"location": "/algorithms/other/dfp/index.html",
"text": "Direct Future Prediction\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nLearning to Act by Predicting the Future\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\n\n\nThe current states (observations and measurements) and the corresponding goal vector are passed as an input to the network. The output of the network is the predicted future measurements for time-steps \nt+1,t+2,t+4,t+8,t+16\n and \nt+32\n for each possible action. \n\n\nFor each action, the measurements of each predicted time-step are multiplied by the goal vector, and the result is a single vector of future values for each action. \n\n\nThen, a weighted sum of the future values of each action is calculated, and the result is a single value for each action. \n\n\nThe action values are passed to the exploration policy to decide on the action to use.\n\n\n\n\nTraining the network\n\n\nGiven a batch of transitions, run them through the network to get the current predictions of the future measurements per action, and set them as the initial targets for training the network. For each transition \n(s_t,a_t,r_t,s_{t+1} )\n in the batch, the target of the network for the action that was taken, is the actual measurements that were seen in time-steps \nt+1,t+2,t+4,t+8,t+16\n and \nt+32\n. For the actions that were not taken, the targets are the current values.",
"title": "Direct Future Prediction"
},
{
"location": "/algorithms/other/dfp/index.html#direct-future-prediction",
"text": "Actions space: Discrete References: Learning to Act by Predicting the Future",
"title": "Direct Future Prediction"
},
{
"location": "/algorithms/other/dfp/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/other/dfp/index.html#algorithm-description",
"text": "Choosing an action The current states (observations and measurements) and the corresponding goal vector are passed as an input to the network. The output of the network is the predicted future measurements for time-steps t+1,t+2,t+4,t+8,t+16 and t+32 for each possible action. For each action, the measurements of each predicted time-step are multiplied by the goal vector, and the result is a single vector of future values for each action. Then, a weighted sum of the future values of each action is calculated, and the result is a single value for each action. The action values are passed to the exploration policy to decide on the action to use. Training the network Given a batch of transitions, run them through the network to get the current predictions of the future measurements per action, and set them as the initial targets for training the network. For each transition (s_t,a_t,r_t,s_{t+1} ) in the batch, the target of the network for the action that was taken, is the actual measurements that were seen in time-steps t+1,t+2,t+4,t+8,t+16 and t+32 . For the actions that were not taken, the targets are the current values.",
"title": "Algorithm Description"
},
{
"location": "/algorithms/imitation/bc/index.html",
"text": "Behavioral Cloning\n\n\nActions space:\n Discrete|Continuous\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\nThe replay buffer contains the expert demonstrations for the task.\nThese demonstrations are given as state, action tuples, and with no reward.\nThe training goal is to reduce the difference between the actions predicted by the network and the actions taken by the expert for each state.\n\n\n\n\nSample a batch of transitions from the replay buffer.\n\n\nUse the current states as input to the network, and the expert actions as the targets of the network.\n\n\nThe loss function for the network is MSE, and therefore we use the Q head to minimize this loss.",
"title": "Behavioral Cloning"
},
{
"location": "/algorithms/imitation/bc/index.html#behavioral-cloning",
"text": "Actions space: Discrete|Continuous",
"title": "Behavioral Cloning"
},
{
"location": "/algorithms/imitation/bc/index.html#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/imitation/bc/index.html#algorithm-description",
"text": "Training the network The replay buffer contains the expert demonstrations for the task.\nThese demonstrations are given as state, action tuples, and with no reward.\nThe training goal is to reduce the difference between the actions predicted by the network and the actions taken by the expert for each state. Sample a batch of transitions from the replay buffer. Use the current states as input to the network, and the expert actions as the targets of the network. The loss function for the network is MSE, and therefore we use the Q head to minimize this loss.",
"title": "Algorithm Description"
},
{
"location": "/dashboard/index.html",
"text": "Reinforcement learning algorithms are neat. That is - when they work. But when they don't, RL algorithms are often quite tricky to debug. \n\n\nFinding the root cause for why things break in RL is rather difficult. Moreover, different RL algorithms shine in some aspects, but then lack on other. Comparing the algorithms faithfully is also a hard task, which requires the right tools.\n\n\nCoach Dashboard is a visualization tool which simplifies the analysis of the training process. Each run of Coach extracts a lot of information from within the algorithm and stores it in the experiment directory. This information is very valuable for debugging, analyzing and comparing different algorithms. But without a good visualization tool, this information can not be utilized. This is where Coach Dashboard takes place.\n\n\nVisualizing Signals\n\n\nCoach Dashboard exposes a convenient user interface for visualizing the training signals. The signals are dynamically updated - during the agent training. Additionaly, it allows selecting a subset of the available signals, and then overlaying them on top of each other. \n\n\n\n\n\n\n\n\n\n\n\n\n\nHolding the CTRL key, while selecting signals, will allow visualizing more than one signal. \n\n\nSignals can be visualized, using either of the Y-axes, in order to visualize signals with different scales. To move a signal to the second Y-axis, select it and press the 'Toggle Second Axis' button.\n\n\n\n\nTracking Statistics\n\n\nWhen running parallel algorithms, such as A3C, it often helps visualizing the learning of all the workers, at the same time. Coach Dashboard allows viewing multiple signals (and even smooth them out, if required) from multiple workers. In addition, it supports viewing the mean and standard deviation of the same signal, across different workers, using Bollinger bands. \n\n\n\n\n\n\n\n\n\n \n\n \nDisplaying Bollinger Bands\n\n\n\n\n\n \n\n \nDisplaying All The Workers\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\nComparing Runs\n\n\nReinforcement learning algorithms are notoriously known as unstable, and suffer from high run-to-run variance. This makes benchmarking and comparing different algorithms even harder. To ease this process, it is common to execute several runs of the same algorithm and average over them. This is easy to do with Coach Dashboard, by centralizing all the experiment directories in a single directory, and then loading them as a single group. Loading several groups of different algorithms then allows comparing the averaged signals, such as the total episode reward. \n\n\nIn RL, there are several interesting performance metrics to consider, and this is easy to do by controlling the X-axis units in Coach Dashboard. It is possible to switch between several options such as the total number of steps or the total training time.\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\nComparing Several Algorithms According to the Time Passed\n\n\n\n\n\n\n\n\n\n\n\n\n\nComparing Several Algorithms According to the Number of Episodes Played",
"title": "Coach Dashboard"
},
{
"location": "/contributing/add_agent/index.html",
"text": "Coach's modularity makes adding an agent a simple and clean task, that involves the following steps:\n\n\n\n\n\n\nImplement your algorithm in a new file under the agents directory. The agent can inherit base classes such as \nValueOptimizationAgent\n or \nActorCriticAgent\n, or the more generic \nAgent\n base class.\n\n\n\n\n\n\nValueOptimizationAgent\n, \nPolicyOptimizationAgent\n and \nAgent\n are abstract classes. \nlearn_from_batch() should be overriden with the desired behavior for the algorithm being implemented. If deciding to inherit from \nAgent\n, also choose_action() should be overriden. \n\n\ndef learn_from_batch(self, batch):\n \"\"\"\n Given a batch of transitions, calculates their target values and updates the network.\n :param batch: A list of transitions\n :return: The loss of the training\n \"\"\"\n pass\n\ndef choose_action(self, curr_state, phase=RunPhase.TRAIN):\n \"\"\"\n choose an action to act with in the current episode being played. Different behavior might be exhibited when training\n or testing.\n\n :param curr_state: the current state to act upon. \n :param phase: the current phase: training or testing.\n :return: chosen action, some action value describing the action (q-value, probability, etc)\n \"\"\"\n pass\n\n\n\n\n\n\n\nMake sure to add your new agent to \nagents/__init__.py\n\n\n\n\n\n\n\n\n\n\nImplement your agent's specific network head, if needed, at the implementation for the framework of your choice. For example \narchitectures/neon_components/heads.py\n. The head will inherit the generic base class Head.\n A new output type should be added to configurations.py, and a mapping between the new head and output type should be defined in the get_output_head() function at \narchitectures/neon_components/general_network.py\n\n\n\n\nDefine a new configuration class at configurations.py, which includes the new agent name in the \ntype\n field, the new output type in the \noutput_types\n field, and assigning default values to hyperparameters.\n\n\n(Optional) Define a preset using the new agent type with a given environment, and the hyperparameters that should be used for training on that environment.",
"title": "Adding a New Agent"
},
{
"location": "/contributing/add_env/index.html",
"text": "Adding a new environment to Coach is as easy as solving CartPole. \n\n\nThere are a few simple steps to follow, and we will walk through them one by one.\n\n\n\n\n\n\nCoach defines a simple API for implementing a new environment which is defined in environment/environment_wrapper.py.\n There are several functions to implement, but only some of them are mandatory. \n\n\nHere are the important ones:\n\n\n def _take_action(self, action_idx):\n \"\"\"\n An environment dependent function that sends an action to the simulator.\n :param action_idx: the action to perform on the environment.\n :return: None\n \"\"\"\n pass\n\n def _preprocess_observation(self, observation):\n \"\"\"\n Do initial observation preprocessing such as cropping, rgb2gray, rescale etc.\n Implementing this function is optional.\n :param observation: a raw observation from the environment\n :return: the preprocessed observation\n \"\"\"\n return observation\n\n def _update_state(self):\n \"\"\"\n Updates the state from the environment.\n Should update self.observation, self.reward, self.done, self.measurements and self.info\n :return: None\n \"\"\"\n pass\n\n def _restart_environment_episode(self, force_environment_reset=False):\n \"\"\"\n :param force_environment_reset: Force the environment to reset even if the episode is not done yet.\n :return:\n \"\"\"\n pass\n\n def get_rendered_image(self):\n \"\"\"\n Return a numpy array containing the image that will be rendered to the screen.\n This can be different from the observation. For example, mujoco's observation is a measurements vector.\n :return: numpy array containing the image that will be rendered to the screen\n \"\"\"\n return self.observation\n\n\n\n\n\n\n\nMake sure to import the environment in environments/__init__.py:\n\n\nfrom doom_environment_wrapper import *\n\n\n\nAlso, a new entry should be added to the EnvTypes enum mapping the environment name to the wrapper's class name:\n\n\nDoom = \"DoomEnvironmentWrapper\"\n\n\n\n\n\n\n\nIn addition a new configuration class should be implemented for defining the environment's parameters and placed in configurations.py. \nFor instance, the following is used for Doom:\n\n\nclass Doom(EnvironmentParameters):\n type = 'Doom'\n frame_skip = 4\n observation_stack_size = 3\n desired_observation_height = 60\n desired_observation_width = 76\n\n\n\n\n\n\n\nAnd that's it, you're done. Now just add a new preset with your newly created environment, and start training an agent on top of it.",
"title": "Adding a New Environment"
}
]
}
+137 -189
View File
@@ -3,31 +3,22 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="./img/favicon.ico">
<title>Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="./css/theme.css" type="text/css" />
<link rel="stylesheet" href="./css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="./css/highlight.css">
<link href="./extra.css" rel="stylesheet">
<script src="./js/jquery-2.1.1.min.js"></script>
<script src="./js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="./js/highlight.pack.js"></script>
<script src="./js/theme.js"></script>
<script>var base_url = '.';</script>
<script data-main="./mkdocs/js/search.js" src="./mkdocs/js/require.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="./js/highlight.pack.js"></script>
</head>
@@ -38,7 +29,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="./index.html" class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href="." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="./search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -47,184 +38,136 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="./index.html">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href=".">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="design/index.html">Design</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="usage/">Usage</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="usage/index.html">Usage</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/dqn/index.html">DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="algorithms/imitation/bc/index.html">Behavioral Cloning</a>
</li>
<li class="">
<a class="" href="design/features/">Features</a>
</li>
<li class="">
<a class="" href="design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="design/network/">Network</a>
</li>
<li class="">
<a class="" href="design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="contributing/add_agent/index.html">Adding a New Agent</a>
</li>
<li class="toctree-l1 ">
<a class="" href="contributing/add_env/index.html">Adding a New Environment</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
<li>
</li>
<li class="toctree-l1">
<a class="" href="dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -236,7 +179,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="./index.html">Reinforcement Learning Coach Documentation</a>
<a href=".">Reinforcement Learning Coach</a>
</nav>
@@ -244,7 +187,7 @@
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="./index.html">Docs</a> &raquo;</li>
<li><a href=".">Docs</a> &raquo;</li>
<li class="wy-breadcrumbs-aside">
@@ -264,8 +207,8 @@
<input name="q" id="mkdocs-search-query" type="text" class="search_input search-query ui-autocomplete-input" placeholder="Search the Docs" autocomplete="off" autofocus>
</form>
<div id="mkdocs-search-results">
Sorry, page not found.
<div id="mkdocs-search-results" class="search-results">
Searching...
</div>
@@ -283,7 +226,7 @@
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -291,13 +234,18 @@
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
</span>
</div>
<script>var base_url = '.';</script>
<script src="./js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="./search/require.js"></script>
<script src="./search/search.js"></script>
</body>
</html>
+7
View File
@@ -0,0 +1,7 @@
/**
* lunr - http://lunrjs.com - A bit like Solr, but much smaller and not as bright - 0.7.0
* Copyright (C) 2016 Oliver Nightingale
* MIT Licensed
* @license
*/
!function(){var t=function(e){var n=new t.Index;return n.pipeline.add(t.trimmer,t.stopWordFilter,t.stemmer),e&&e.call(n,n),n};t.version="0.7.0",t.utils={},t.utils.warn=function(t){return function(e){t.console&&console.warn&&console.warn(e)}}(this),t.utils.asString=function(t){return void 0===t||null===t?"":t.toString()},t.EventEmitter=function(){this.events={}},t.EventEmitter.prototype.addListener=function(){var t=Array.prototype.slice.call(arguments),e=t.pop(),n=t;if("function"!=typeof e)throw new TypeError("last argument must be a function");n.forEach(function(t){this.hasHandler(t)||(this.events[t]=[]),this.events[t].push(e)},this)},t.EventEmitter.prototype.removeListener=function(t,e){if(this.hasHandler(t)){var n=this.events[t].indexOf(e);this.events[t].splice(n,1),this.events[t].length||delete this.events[t]}},t.EventEmitter.prototype.emit=function(t){if(this.hasHandler(t)){var e=Array.prototype.slice.call(arguments,1);this.events[t].forEach(function(t){t.apply(void 0,e)})}},t.EventEmitter.prototype.hasHandler=function(t){return t in this.events},t.tokenizer=function(e){return arguments.length&&null!=e&&void 0!=e?Array.isArray(e)?e.map(function(e){return t.utils.asString(e).toLowerCase()}):e.toString().trim().toLowerCase().split(t.tokenizer.seperator):[]},t.tokenizer.seperator=/[\s\-]+/,t.tokenizer.load=function(t){var e=this.registeredFunctions[t];if(!e)throw new Error("Cannot load un-registered function: "+t);return e},t.tokenizer.label="default",t.tokenizer.registeredFunctions={"default":t.tokenizer},t.tokenizer.registerFunction=function(e,n){n in this.registeredFunctions&&t.utils.warn("Overwriting existing tokenizer: "+n),e.label=n,this.registeredFunctions[n]=e},t.Pipeline=function(){this._stack=[]},t.Pipeline.registeredFunctions={},t.Pipeline.registerFunction=function(e,n){n in this.registeredFunctions&&t.utils.warn("Overwriting existing registered function: "+n),e.label=n,t.Pipeline.registeredFunctions[e.label]=e},t.Pipeline.warnIfFunctionNotRegistered=function(e){var n=e.label&&e.label in this.registeredFunctions;n||t.utils.warn("Function is not registered with pipeline. This may cause problems when serialising the index.\n",e)},t.Pipeline.load=function(e){var n=new t.Pipeline;return e.forEach(function(e){var i=t.Pipeline.registeredFunctions[e];if(!i)throw new Error("Cannot load un-registered function: "+e);n.add(i)}),n},t.Pipeline.prototype.add=function(){var e=Array.prototype.slice.call(arguments);e.forEach(function(e){t.Pipeline.warnIfFunctionNotRegistered(e),this._stack.push(e)},this)},t.Pipeline.prototype.after=function(e,n){t.Pipeline.warnIfFunctionNotRegistered(n);var i=this._stack.indexOf(e);if(-1==i)throw new Error("Cannot find existingFn");i+=1,this._stack.splice(i,0,n)},t.Pipeline.prototype.before=function(e,n){t.Pipeline.warnIfFunctionNotRegistered(n);var i=this._stack.indexOf(e);if(-1==i)throw new Error("Cannot find existingFn");this._stack.splice(i,0,n)},t.Pipeline.prototype.remove=function(t){var e=this._stack.indexOf(t);-1!=e&&this._stack.splice(e,1)},t.Pipeline.prototype.run=function(t){for(var e=[],n=t.length,i=this._stack.length,r=0;n>r;r++){for(var o=t[r],s=0;i>s&&(o=this._stack[s](o,r,t),void 0!==o&&""!==o);s++);void 0!==o&&""!==o&&e.push(o)}return e},t.Pipeline.prototype.reset=function(){this._stack=[]},t.Pipeline.prototype.toJSON=function(){return this._stack.map(function(e){return t.Pipeline.warnIfFunctionNotRegistered(e),e.label})},t.Vector=function(){this._magnitude=null,this.list=void 0,this.length=0},t.Vector.Node=function(t,e,n){this.idx=t,this.val=e,this.next=n},t.Vector.prototype.insert=function(e,n){this._magnitude=void 0;var i=this.list;if(!i)return this.list=new t.Vector.Node(e,n,i),this.length++;if(e<i.idx)return this.list=new t.Vector.Node(e,n,i),this.length++;for(var r=i,o=i.next;void 0!=o;){if(e<o.idx)return r.next=new t.Vector.Node(e,n,o),this.length++;r=o,o=o.next}return r.next=new t.Vector.Node(e,n,o),this.length++},t.Vector.prototype.magnitude=function(){if(this._magnitude)return this._magnitude;for(var t,e=this.list,n=0;e;)t=e.val,n+=t*t,e=e.next;return this._magnitude=Math.sqrt(n)},t.Vector.prototype.dot=function(t){for(var e=this.list,n=t.list,i=0;e&&n;)e.idx<n.idx?e=e.next:e.idx>n.idx?n=n.next:(i+=e.val*n.val,e=e.next,n=n.next);return i},t.Vector.prototype.similarity=function(t){return this.dot(t)/(this.magnitude()*t.magnitude())},t.SortedSet=function(){this.length=0,this.elements=[]},t.SortedSet.load=function(t){var e=new this;return e.elements=t,e.length=t.length,e},t.SortedSet.prototype.add=function(){var t,e;for(t=0;t<arguments.length;t++)e=arguments[t],~this.indexOf(e)||this.elements.splice(this.locationFor(e),0,e);this.length=this.elements.length},t.SortedSet.prototype.toArray=function(){return this.elements.slice()},t.SortedSet.prototype.map=function(t,e){return this.elements.map(t,e)},t.SortedSet.prototype.forEach=function(t,e){return this.elements.forEach(t,e)},t.SortedSet.prototype.indexOf=function(t){for(var e=0,n=this.elements.length,i=n-Line truncated
File renamed without changes.
File renamed without changes.
@@ -1,8 +1,12 @@
require.config({
baseUrl: base_url + "/search/"
});
require([
base_url + '/mkdocs/js/mustache.min.js',
base_url + '/mkdocs/js/lunr-0.5.7.min.js',
'mustache.min',
'lunr.min',
'text!search-results-template.mustache',
'text!../search_index.txt',
'text!search_index.json',
], function (Mustache, lunr, results_template, data) {
"use strict";
@@ -70,7 +74,7 @@ require([
*/
jQuery('#mkdocs_search_modal a').click(function(){
jQuery('#mkdocs_search_modal').modal('hide');
})
});
}
};
@@ -83,6 +87,6 @@ require([
search();
}
search_input.addEventListener("keyup", search);
if (search_input){search_input.addEventListener("keyup", search);}
});
+704
View File
@@ -0,0 +1,704 @@
{
"docs": [
{
"location": "/",
"text": "What is Coach?\n\n\nMotivation\n\n\nTrain and evaluate reinforcement learning agents by harnessing the power of multi-core CPU processing to achieve state-of-the-art results. Provide a sandbox for easing the development process of new algorithms through a modular design and an elegant set of APIs. \n\n\nSolution\n\n\nCoach is a python environment which models the interaction between an agent and an environment in a modular way.\nWith Coach, it is possible to model an agent by combining various building blocks, and training the agent on multiple environments.\nThe available environments allow testing the agent in different practical fields such as robotics, autonomous driving, games and more. \nCoach collects statistics from the training process and supports advanced visualization techniques for debugging the agent being trained.\n\n\nBlog post from the Intel\u00ae AI website can be found \nhere\n.\n\n\nGitHub repository is \nhere\n. \n\n\nDesign",
"title": "Home"
},
{
"location": "/#what-is-coach",
"text": "",
"title": "What is Coach?"
},
{
"location": "/#motivation",
"text": "Train and evaluate reinforcement learning agents by harnessing the power of multi-core CPU processing to achieve state-of-the-art results. Provide a sandbox for easing the development process of new algorithms through a modular design and an elegant set of APIs.",
"title": "Motivation"
},
{
"location": "/#solution",
"text": "Coach is a python environment which models the interaction between an agent and an environment in a modular way.\nWith Coach, it is possible to model an agent by combining various building blocks, and training the agent on multiple environments.\nThe available environments allow testing the agent in different practical fields such as robotics, autonomous driving, games and more. \nCoach collects statistics from the training process and supports advanced visualization techniques for debugging the agent being trained. Blog post from the Intel\u00ae AI website can be found here . GitHub repository is here .",
"title": "Solution"
},
{
"location": "/#design",
"text": "",
"title": "Design"
},
{
"location": "/usage/",
"text": "Coach Usage\n\n\nTraining an Agent\n\n\nSingle-threaded Algorithms\n\n\nThis is the most common case. Just choose a preset using the \n-p\n flag and press enter.\n\n\nExample:\n\n\npython coach.py -p CartPole_DQN\n\n\nMulti-threaded Algorithms\n\n\nMulti-threaded algorithms are very common this days.\nThey typically achieve the best results, and scale gracefully with the number of threads.\nIn Coach, running such algorithms is done by selecting a suitable preset, and choosing the number of threads to run using the \n-n\n flag.\n\n\nExample:\n\n\npython coach.py -p CartPole_A3C -n 8\n\n\nEvaluating an Agent\n\n\nThere are several options for evaluating an agent during the training:\n\n\n\n\n\n\nFor multi-threaded runs, an evaluation agent will constantly run in the background and evaluate the model during the training.\n\n\n\n\n\n\nFor single-threaded runs, it is possible to define an evaluation period through the preset. This will run several episodes of evaluation once in a while.\n\n\n\n\n\n\nAdditionally, it is possible to save checkpoints of the agents networks and then run only in evaluation mode.\nSaving checkpoints can be done by specifying the number of seconds between storing checkpoints using the \n-s\n flag.\nThe checkpoints will be saved into the experiment directory.\nLoading a model for evaluation can be done by specifying the \n-crd\n flag with the experiment directory, and the \n--evaluate\n flag to disable training.\n\n\nExample:\n\n\npython coach.py -p CartPole_DQN -s 60\n\n\npython coach.py -p CartPole_DQN --evaluate -crd CHECKPOINT_RESTORE_DIR\n\n\nPlaying with the Environment as a Human\n\n\nInteracting with the environment as a human can be useful for understanding its difficulties and for collecting data for imitation learning.\nIn Coach, this can be easily done by selecting a preset that defines the environment to use, and specifying the \n--play\n flag.\nWhen the environment is loaded, the available keyboard buttons will be printed to the screen.\nPressing the escape key when finished will end the simulation and store the replay buffer in the experiment dir.\n\n\nExample:\n\n\npython coach.py -p Breakout_DQN --play\n\n\nLearning Through Imitation Learning\n\n\nLearning through imitation of human behavior is a nice way to speedup the learning.\nIn Coach, this can be done in two steps -\n\n\n\n\n\n\nCreate a dataset of demonstrations by playing with the environment as a human.\n After this step, a pickle of the replay buffer containing your game play will be stored in the experiment directory.\n The path to this replay buffer will be printed to the screen.\n To do so, you should select an environment type and level through the command line, and specify the \n--play\n flag.\n\n\nExample:\n\n\npython coach.py -et Doom -lvl Basic --play\n\n\n\n\n\n\nNext, use an imitation learning preset and set the replay buffer path accordingly.\n The path can be set either from the command line or from the preset itself.\n\n\nExample:\n\n\npython coach.py -p Doom_Basic_BC -cp='agent.load_memory_from_file_path=\\\"<experiment dir>/replay_buffer.p\\\"'\n\n\n\n\n\n\nVisualizations\n\n\nRendering the Environment\n\n\nRendering the environment can be done by using the \n-r\n flag.\nWhen working with multi-threaded algorithms, the rendered image will be representing the game play of the evaluation worker.\nWhen working with single-threaded algorithms, the rendered image will be representing the single worker which can be either training or evaluating.\nKeep in mind that rendering the environment in single-threaded algorithms may slow the training to some extent.\nWhen playing with the environment using the \n--play\n flag, the environment will be rendered automatically without the need for specifying the \n-r\n flag.\n\n\nExample:\n\n\npython coach.py -p Breakout_DQN -r\n\n\nDumping GIFs\n\n\nCoach allows storing GIFs of the agent game play.\nTo dump GIF files, use the \n-dg\n flag.\nThe files are dumped after every evaluation episode, and are saved into the experiment directory, under a gifs sub-directory.\n\n\nExample:\n\n\npython coach.py -p Breakout_A3C -n 4 -dg\n\n\nSwitching between deep learning frameworks\n\n\nCoach uses TensorFlow as its main backend framework, but it also supports neon for some of the algorithms.\nBy default, TensorFlow will be used. It is possible to switch to neon using the \n-f\n flag.\n\n\nExample:\n\n\npython coach.py -p Doom_Basic_DQN -f neon\n\n\nAdditional Flags\n\n\nThere are several convenient flags which are important to know about.\nHere we will list most of the flags, but these can be updated from time to time.\nThe most up to date description can be found by using the \n-h\n flag.\n\n\n\n\n\n\n\n\nFlag\n\n\nType\n\n\nDescription\n\n\n\n\n\n\n\n\n\n\n-p PRESET\n, \n`--preset PRESET\n\n\nstring\n\n\nName of a preset to run (as configured in presets.py)\n\n\n\n\n\n\n-l\n, \n--list\n\n\nflag\n\n\nList all available presets\n\n\n\n\n\n\n-e ELine truncated
"title": "Usage"
},
{
"location": "/usage/#coach-usage",
"text": "",
"title": "Coach Usage"
},
{
"location": "/usage/#training-an-agent",
"text": "",
"title": "Training an Agent"
},
{
"location": "/usage/#single-threaded-algorithms",
"text": "This is the most common case. Just choose a preset using the -p flag and press enter. Example: python coach.py -p CartPole_DQN",
"title": "Single-threaded Algorithms"
},
{
"location": "/usage/#multi-threaded-algorithms",
"text": "Multi-threaded algorithms are very common this days.\nThey typically achieve the best results, and scale gracefully with the number of threads.\nIn Coach, running such algorithms is done by selecting a suitable preset, and choosing the number of threads to run using the -n flag. Example: python coach.py -p CartPole_A3C -n 8",
"title": "Multi-threaded Algorithms"
},
{
"location": "/usage/#evaluating-an-agent",
"text": "There are several options for evaluating an agent during the training: For multi-threaded runs, an evaluation agent will constantly run in the background and evaluate the model during the training. For single-threaded runs, it is possible to define an evaluation period through the preset. This will run several episodes of evaluation once in a while. Additionally, it is possible to save checkpoints of the agents networks and then run only in evaluation mode.\nSaving checkpoints can be done by specifying the number of seconds between storing checkpoints using the -s flag.\nThe checkpoints will be saved into the experiment directory.\nLoading a model for evaluation can be done by specifying the -crd flag with the experiment directory, and the --evaluate flag to disable training. Example: python coach.py -p CartPole_DQN -s 60 python coach.py -p CartPole_DQN --evaluate -crd CHECKPOINT_RESTORE_DIR",
"title": "Evaluating an Agent"
},
{
"location": "/usage/#playing-with-the-environment-as-a-human",
"text": "Interacting with the environment as a human can be useful for understanding its difficulties and for collecting data for imitation learning.\nIn Coach, this can be easily done by selecting a preset that defines the environment to use, and specifying the --play flag.\nWhen the environment is loaded, the available keyboard buttons will be printed to the screen.\nPressing the escape key when finished will end the simulation and store the replay buffer in the experiment dir. Example: python coach.py -p Breakout_DQN --play",
"title": "Playing with the Environment as a Human"
},
{
"location": "/usage/#learning-through-imitation-learning",
"text": "Learning through imitation of human behavior is a nice way to speedup the learning.\nIn Coach, this can be done in two steps - Create a dataset of demonstrations by playing with the environment as a human.\n After this step, a pickle of the replay buffer containing your game play will be stored in the experiment directory.\n The path to this replay buffer will be printed to the screen.\n To do so, you should select an environment type and level through the command line, and specify the --play flag. Example: python coach.py -et Doom -lvl Basic --play Next, use an imitation learning preset and set the replay buffer path accordingly.\n The path can be set either from the command line or from the preset itself. Example: python coach.py -p Doom_Basic_BC -cp='agent.load_memory_from_file_path=\\\"<experiment dir>/replay_buffer.p\\\"'",
"title": "Learning Through Imitation Learning"
},
{
"location": "/usage/#visualizations",
"text": "",
"title": "Visualizations"
},
{
"location": "/usage/#rendering-the-environment",
"text": "Rendering the environment can be done by using the -r flag.\nWhen working with multi-threaded algorithms, the rendered image will be representing the game play of the evaluation worker.\nWhen working with single-threaded algorithms, the rendered image will be representing the single worker which can be either training or evaluating.\nKeep in mind that rendering the environment in single-threaded algorithms may slow the training to some extent.\nWhen playing with the environment using the --play flag, the environment will be rendered automatically without the need for specifying the -r flag. Example: python coach.py -p Breakout_DQN -r",
"title": "Rendering the Environment"
},
{
"location": "/usage/#dumping-gifs",
"text": "Coach allows storing GIFs of the agent game play.\nTo dump GIF files, use the -dg flag.\nThe files are dumped after every evaluation episode, and are saved into the experiment directory, under a gifs sub-directory. Example: python coach.py -p Breakout_A3C -n 4 -dg",
"title": "Dumping GIFs"
},
{
"location": "/usage/#switching-between-deep-learning-frameworks",
"text": "Coach uses TensorFlow as its main backend framework, but it also supports neon for some of the algorithms.\nBy default, TensorFlow will be used. It is possible to switch to neon using the -f flag. Example: python coach.py -p Doom_Basic_DQN -f neon",
"title": "Switching between deep learning frameworks"
},
{
"location": "/usage/#additional-flags",
"text": "There are several convenient flags which are important to know about.\nHere we will list most of the flags, but these can be updated from time to time.\nThe most up to date description can be found by using the -h flag. Flag Type Description -p PRESET , `--preset PRESET string Name of a preset to run (as configured in presets.py) -l , --list flag List all available presets -e EXPERIMENT_NAME , --experiment_name EXPERIMENT_NAME string Experiment name to be used to store the results. -r , --render flag Render environment -f FRAMEWORK , --framework FRAMEWORK string Neural network framework. Available values: tensorflow, neon -n NUM_WORKERS , --num_workers NUM_WORKERS int Number of workers for multi-process based agents, e.g. A3C --play flag Play as a human by controlling the game with the keyboard. This option will save a replay buffer with the game play. --evaluate flag Run evaluation only. This is a convenient way to disable training in order to evaluate an existing checkpoint. -v , --verbose flag Don't suppress TensorFlow debug prints. -s SAVE_MODEL_SEC , --save_model_sec SAVE_MODEL_SEC int Time in seconds between saving checkpoints of the model. -crd CHECKPOINT_RESTORE_DIR , --checkpoint_restore_dir CHECKPOINT_RESTORE_DIR string Path to a folder containing a checkpoint to restore the model from. -dg , --dump_gifs flag Enable the gif saving functionality. -at AGENT_TYPE , --agent_type AGENT_TYPE string Choose an agent type class to override on top of the selected preset. If no preset is defined, a preset can be set from the command-line by combining settings which are set by using --agent_type , --experiment_type , --environemnt_type -et ENVIRONMENT_TYPE , --environment_type ENVIRONMENT_TYPE string Choose an environment type class to override on top of the selected preset. If no preset is defined, a preset can be set from the command-line by combining settings which are set by using --agent_type , --experiment_type , --environemnt_type -ept EXPLORATION_POLICY_TYPE , --exploration_policy_type EXPLORATION_POLICY_TYPE string Choose an exploration policy type class to override on top of the selected preset.If no preset is defined, a preset can be set from the command-line by combining settings which are set by using --agent_type , --experiment_type , --environemnt_type -lvl LEVEL , --level LEVEL string Choose the level that will be played in the environment that was selected. This value will override the level parameter in the environment class. -cp CUSTOM_PARAMETER , --custom_parameter CUSTOM_PARAMETER string Semicolon separated parameters used to override specific parameters on top of the selected preset (or on top of the command-line assembled one). Whenever a parameter value is a string, it should be inputted as '\\\"string\\\"' . For ex.: \"visualization.render=False; num_training_iterations=500; optimizer='rmsprop'\"",
"title": "Additional Flags"
},
{
"location": "/design/features/",
"text": "Coach Features\n\n\nSupported Algorithms\n\n\nCoach supports many state-of-the-art reinforcement learning algorithms, which are separated into two main classes -\nvalue optimization and policy optimization. A detailed description of those algorithms may be found in the algorithms\nsection.\n\n\n\n\n\n\n\n\n\n\n\nSupported Environments\n\n\nCoach supports a large number of environments which can be solved using reinforcement learning:\n\n\n\n\n\n\nDeepMind Control Suite\n - a set of reinforcement learning environments\n powered by the MuJoCo physics engine.\n\n\n\n\n\n\nBlizzard Starcraft II\n - a popular strategy game which was wrapped with a\n python interface by DeepMind.\n\n\n\n\n\n\nViZDoom\n - a Doom-based AI research platform for reinforcement learning\n from raw visual information.\n\n\n\n\n\n\nCARLA\n - an open-source simulator for autonomous driving research.\n\n\n\n\n\n\nOpenAI Gym\n - a library which consists of a set of environments, from games to robotics.\n Additionally, it can be extended using the API defined by the authors.\n\n\n\n\n\n\nIn Coach, we support all the native environments in Gym, along with several extensions such as:\n\n\n\n\n\n\nRoboschool\n - a set of environments powered by the PyBullet engine,\n that offer a free alternative to MuJoCo.\n\n\n\n\n\n\nGym Extensions\n - a set of environments that extends Gym for\n auxiliary tasks (multitask learning, transfer learning, inverse reinforcement learning, etc.)\n\n\n\n\n\n\nPyBullet\n - a physics engine that\n includes a set of robotics environments.",
"title": "Features"
},
{
"location": "/design/features/#coach-features",
"text": "",
"title": "Coach Features"
},
{
"location": "/design/features/#supported-algorithms",
"text": "Coach supports many state-of-the-art reinforcement learning algorithms, which are separated into two main classes -\nvalue optimization and policy optimization. A detailed description of those algorithms may be found in the algorithms\nsection.",
"title": "Supported Algorithms"
},
{
"location": "/design/features/#supported-environments",
"text": "Coach supports a large number of environments which can be solved using reinforcement learning: DeepMind Control Suite - a set of reinforcement learning environments\n powered by the MuJoCo physics engine. Blizzard Starcraft II - a popular strategy game which was wrapped with a\n python interface by DeepMind. ViZDoom - a Doom-based AI research platform for reinforcement learning\n from raw visual information. CARLA - an open-source simulator for autonomous driving research. OpenAI Gym - a library which consists of a set of environments, from games to robotics.\n Additionally, it can be extended using the API defined by the authors. In Coach, we support all the native environments in Gym, along with several extensions such as: Roboschool - a set of environments powered by the PyBullet engine,\n that offer a free alternative to MuJoCo. Gym Extensions - a set of environments that extends Gym for\n auxiliary tasks (multitask learning, transfer learning, inverse reinforcement learning, etc.) PyBullet - a physics engine that\n includes a set of robotics environments.",
"title": "Supported Environments"
},
{
"location": "/design/control_flow/",
"text": "Coach Control Flow\n\n\nCoach is built in a modular way, encouraging modules reuse and reducing the amount of boilerplate code needed\nfor developing new algorithms or integrating a new challenge as an environment.\nOn the other hand, it can be overwhelming for new users to ramp up on the code.\nTo help with that, here's a short overview of the control flow.\n\n\nGraph Manager\n\n\nThe main entry point for Coach is \ncoach.py\n.\nThe main functionality of this script is to parse the command line arguments and invoke all the sub-processes needed\nfor the given experiment.\n\ncoach.py\n executes the given \npreset\n file which returns a \nGraphManager\n object.\n\n\nA \npreset\n is a design pattern that is intended for concentrating the entire definition of an experiment in a single\nfile. This helps with experiments reproducibility, improves readability and prevents confusion.\nThe outcome of a preset is a \nGraphManager\n which will usually be instantiated in the final lines of the preset.\n\n\nA \nGraphManager\n is an object that holds all the agents and environments of an experiment, and is mostly responsible\nfor scheduling their work. Why is it called a \ngraph\n manager? Because agents and environments are structured into\na graph of interactions. For example, in hierarchical reinforcement learning schemes, there will often be a master\npolicy agent, that will control a sub-policy agent, which will interact with the environment. Other schemes can have\nmuch more complex graphs of control, such as several hierarchy layers, each with multiple agents.\nThe graph manager's main loop is the improve loop.\n\n\n\n\n\n\n\n\n\n\n\nThe improve loop skips between 3 main phases - heatup, training and evaluation:\n\n\n\n\n\n\nHeatup\n - the goal of this phase is to collect initial data for populating the replay buffers. The heatup phase\n takes place only in the beginning of the experiment, and the agents will act completely randomly during this phase.\n Importantly, the agents do not train their networks during this phase. DQN for example, uses 50k random steps in order\n to initialize the replay buffers.\n\n\n\n\n\n\nTraining\n - the training phase is the main phase of the experiment. This phase can change between agent types,\n but essentially consists of repeated cycles of acting, collecting data from the environment, and training the agent\n networks. During this phase, the agent will use its exploration policy in training mode, which will add noise to its\n actions in order to improve its knowledge about the environment state space.\n\n\n\n\n\n\nEvaluation\n - the evaluation phase is intended for evaluating the current performance of the agent. The agents\n will act greedily in order to exploit the knowledge aggregated so far and the performance over multiple episodes of\n evaluation will be averaged in order to reduce the stochasticity effects of all the components.\n\n\n\n\n\n\nLevel Manager\n\n\nIn each of the 3 phases described above, the graph manager will invoke all the hierarchy levels in the graph in a\nsynchronized manner. In Coach, agents do not interact directly with the environment. Instead, they go through a\n\nLevelManager\n, which is a proxy that manages their interaction. The level manager passes the current state and reward\nfrom the environment to the agent, and the actions from the agent to the environment.\n\n\nThe motivation for having a level manager is to disentangle the code of the environment and the agent, so to allow more\ncomplex interactions. Each level can have multiple agents which interact with the environment. Who gets to choose the\naction for each step is controlled by the level manager.\nAdditionally, each level manager can act as an environment for the hierarchy level above it, such that each hierarchy\nlevel can be seen as an interaction between an agent and an environment, even if the environment is just more agents in\na lower hierarchy level.\n\n\nAgent\n\n\nThe base agent class has 3 main function that will be used during those phases - observe, act and train.\n\n\n\n\nObserve\n - this function gets the latest response from the environment as input, and updates the internal state\n of the agent with the new information. The environment response will\n be first passed through the agent's \nInputFilter\n object, which will process the values in the response, according\n to the specific agent definition. The environment response will then be converted into a\n \nTransition\n which will contain the information from a single step\n (\n s_{t}, a_{t}, r_{t}, s_{t+1}, terminal signal \n), and store it in the memory.\n\n\n\n\n\n\n\n\nAct\n - this function uses the current internal state of the agent in order to select the next action to take on\n the environment. This function will call the per-agent custom function \nchoose_action\n that will use the network\n and the exploration policy in order to select an action. The action will be storLine truncated
"title": "Control Flow"
},
{
"location": "/design/control_flow/#coach-control-flow",
"text": "Coach is built in a modular way, encouraging modules reuse and reducing the amount of boilerplate code needed\nfor developing new algorithms or integrating a new challenge as an environment.\nOn the other hand, it can be overwhelming for new users to ramp up on the code.\nTo help with that, here's a short overview of the control flow.",
"title": "Coach Control Flow"
},
{
"location": "/design/control_flow/#graph-manager",
"text": "The main entry point for Coach is coach.py .\nThe main functionality of this script is to parse the command line arguments and invoke all the sub-processes needed\nfor the given experiment. coach.py executes the given preset file which returns a GraphManager object. A preset is a design pattern that is intended for concentrating the entire definition of an experiment in a single\nfile. This helps with experiments reproducibility, improves readability and prevents confusion.\nThe outcome of a preset is a GraphManager which will usually be instantiated in the final lines of the preset. A GraphManager is an object that holds all the agents and environments of an experiment, and is mostly responsible\nfor scheduling their work. Why is it called a graph manager? Because agents and environments are structured into\na graph of interactions. For example, in hierarchical reinforcement learning schemes, there will often be a master\npolicy agent, that will control a sub-policy agent, which will interact with the environment. Other schemes can have\nmuch more complex graphs of control, such as several hierarchy layers, each with multiple agents.\nThe graph manager's main loop is the improve loop. The improve loop skips between 3 main phases - heatup, training and evaluation: Heatup - the goal of this phase is to collect initial data for populating the replay buffers. The heatup phase\n takes place only in the beginning of the experiment, and the agents will act completely randomly during this phase.\n Importantly, the agents do not train their networks during this phase. DQN for example, uses 50k random steps in order\n to initialize the replay buffers. Training - the training phase is the main phase of the experiment. This phase can change between agent types,\n but essentially consists of repeated cycles of acting, collecting data from the environment, and training the agent\n networks. During this phase, the agent will use its exploration policy in training mode, which will add noise to its\n actions in order to improve its knowledge about the environment state space. Evaluation - the evaluation phase is intended for evaluating the current performance of the agent. The agents\n will act greedily in order to exploit the knowledge aggregated so far and the performance over multiple episodes of\n evaluation will be averaged in order to reduce the stochasticity effects of all the components.",
"title": "Graph Manager"
},
{
"location": "/design/control_flow/#level-manager",
"text": "In each of the 3 phases described above, the graph manager will invoke all the hierarchy levels in the graph in a\nsynchronized manner. In Coach, agents do not interact directly with the environment. Instead, they go through a LevelManager , which is a proxy that manages their interaction. The level manager passes the current state and reward\nfrom the environment to the agent, and the actions from the agent to the environment. The motivation for having a level manager is to disentangle the code of the environment and the agent, so to allow more\ncomplex interactions. Each level can have multiple agents which interact with the environment. Who gets to choose the\naction for each step is controlled by the level manager.\nAdditionally, each level manager can act as an environment for the hierarchy level above it, such that each hierarchy\nlevel can be seen as an interaction between an agent and an environment, even if the environment is just more agents in\na lower hierarchy level.",
"title": "Level Manager"
},
{
"location": "/design/control_flow/#agent",
"text": "The base agent class has 3 main function that will be used during those phases - observe, act and train. Observe - this function gets the latest response from the environment as input, and updates the internal state\n of the agent with the new information. The environment response will\n be first passed through the agent's InputFilter object, which will process the values in the response, according\n to the specific agent definition. The environment response will then be converted into a\n Transition which will contain the information from a single step\n ( s_{t}, a_{t}, r_{t}, s_{t+1}, terminal signal ), and store it in the memory. Act - this function uses the current internal state of the agent in order to select the next action to take on\n the environment. This function will call the per-agent custom function choose_action that will use the network\n and the exploration policy in order to select an action. The action will be stored, together with any additional\n information (like the action value for example) in an ActionInfo object. The ActionInfo object will then be\n passed through the agent's OutputFilter to allow any processing of the action (like discretization,\n or shifting, for example), before passing it to the environment. Train - this function will sample a batch from the memory and train on it. The batch of transitions will be\n first wrapped into a Batch object to allow efficient querying of the batch values. It will then be passed into\n the agent specific learn_from_batch function, that will extract network target values from the batch and will\n train the networks accordingly. Lastly, if there's a target network defined for the agent, it will sync the target\n network weights with the online network.",
"title": "Agent"
},
{
"location": "/design/network/",
"text": "Network Design\n\n\nEach agent has at least one neural network, used as the function approximator, for choosing the actions. The network is designed in a modular way to allow reusability in different agents. It is separated into three main parts:\n\n\n\n\n\n\nInput Embedders\n - This is the first stage of the network, meant to convert the input into a feature vector representation. It is possible to combine several instances of any of the supported embedders, in order to allow varied combinations of inputs. \n\n\nThere are two main types of input embedders: \n\n\n\n\nImage embedder - Convolutional neural network. \n\n\nVector embedder - Multi-layer perceptron. \n\n\n\n\n\n\n\n\nMiddlewares\n - The middleware gets the output of the input embedder, and processes it into a different representation domain, before sending it through the output head. The goal of the middleware is to enable processing the combined outputs of several input embedders, and pass them through some extra processing. This, for instance, might include an LSTM or just a plain simple FC layer.\n\n\n\n\n\n\nOutput Heads\n - The output head is used in order to predict the values required from the network. These might include action-values, state-values or a policy. As with the input embedders, it is possible to use several output heads in the same network. For example, the \nActor Critic\n agent combines two heads - a policy head and a state-value head.\n In addition, the output heads defines the loss function according to the head type.\n\n\n\n\n\n\n\u200b\n\n\n\n\n\n\n\n\n\n\n\nKeeping Network Copies in Sync\n\n\nMost of the reinforcement learning agents include more than one copy of the neural network. These copies serve as counterparts of the main network which are updated in different rates, and are often synchronized either locally or between parallel workers. For easier synchronization of those copies, a wrapper around these copies exposes a simplified API, which allows hiding these complexities from the agent.",
"title": "Network"
},
{
"location": "/design/network/#network-design",
"text": "Each agent has at least one neural network, used as the function approximator, for choosing the actions. The network is designed in a modular way to allow reusability in different agents. It is separated into three main parts: Input Embedders - This is the first stage of the network, meant to convert the input into a feature vector representation. It is possible to combine several instances of any of the supported embedders, in order to allow varied combinations of inputs. There are two main types of input embedders: Image embedder - Convolutional neural network. Vector embedder - Multi-layer perceptron. Middlewares - The middleware gets the output of the input embedder, and processes it into a different representation domain, before sending it through the output head. The goal of the middleware is to enable processing the combined outputs of several input embedders, and pass them through some extra processing. This, for instance, might include an LSTM or just a plain simple FC layer. Output Heads - The output head is used in order to predict the values required from the network. These might include action-values, state-values or a policy. As with the input embedders, it is possible to use several output heads in the same network. For example, the Actor Critic agent combines two heads - a policy head and a state-value head.\n In addition, the output heads defines the loss function according to the head type. \u200b",
"title": "Network Design"
},
{
"location": "/design/network/#keeping-network-copies-in-sync",
"text": "Most of the reinforcement learning agents include more than one copy of the neural network. These copies serve as counterparts of the main network which are updated in different rates, and are often synchronized either locally or between parallel workers. For easier synchronization of those copies, a wrapper around these copies exposes a simplified API, which allows hiding these complexities from the agent.",
"title": "Keeping Network Copies in Sync"
},
{
"location": "/design/filters/",
"text": "Filters\n\n\nFilters are a mechanism in Coach that allows doing pre-processing and post-processing of the internal agent information.\nThere are two filter categories -\n\n\n\n\n\n\nInput filters\n - these are filters that process the information passed \ninto\n the agent from the environment.\n This information includes the observation and the reward. Input filters therefore allow rescaling observations,\n normalizing rewards, stack observations, etc.\n\n\n\n\n\n\nOutput filters\n - these are filters that process the information going \nout\n of the agent into the environment.\n This information includes the action the agent chooses to take. Output filters therefore allow conversion of\n actions from one space into another. For example, the agent can take \n N \n discrete actions, that will be mapped by\n the output filter onto \n N \n continuous actions.\n\n\n\n\n\n\nFilters can be stacked on top of each other in order to build complex processing flows of the inputs or outputs.\n\n\n\n\n\n\n\n\n\n\n\nInput Filters\n\n\nThe input filters are separated into two categories - \nobservation filters\n and \nreward filters\n.\n\n\nObservation Filters\n\n\n\n\n\n\nObservationClippingFilter\n - Clips the observation values to a given range of values. For example, if the\n observation consists of measurements in an arbitrary range, and we want to control the minimum and maximum values\n of these observations, we can define a range and clip the values of the measurements.\n\n\n\n\n\n\nObservationCropFilter\n - Crops the size of the observation to a given crop window. For example, in Atari, the\n observations are images with a shape of 210x160. Usually, we will want to crop the size of the observation to a\n square of 160x160 before rescaling them.\n\n\n\n\n\n\nObservationMoveAxisFilter\n - Reorders the axes of the observation. This can be useful when the observation is an\n image, and we want to move the channel axis to be the last axis instead of the first axis.\n\n\n\n\n\n\nObservationNormalizationFilter\n - Normalizes the observation values with a running mean and standard deviation of\n all the observations seen so far. The normalization is performed element-wise. Additionally, when working with\n multiple workers, the statistics used for the normalization operation are accumulated over all the workers.\n\n\n\n\n\n\nObservationReductionBySubPartsNameFilter\n - Allows keeping only parts of the observation, by specifying their\n name. For example, the CARLA environment extracts multiple measurements that can be used by the agent, such as\n speed and location. If we want to only use the speed, it can be done using this filter.\n\n\n\n\n\n\nObservationRescaleSizeByFactorFilter\n - Rescales an image observation by some factor. For example, the image size\n can be reduced by a factor of 2.\n\n\n\n\n\n\nObservationRescaleToSizeFilter\n - Rescales an image observation to a given size. The target size does not\n necessarily keep the aspect ratio of the original observation.\n\n\n\n\n\n\nObservationRGBToYFilter\n - Converts a color image observation specified using the RGB encoding into a grayscale\n image observation, by keeping only the luminance (Y) channel of the YUV encoding. This can be useful if the colors\n in the original image are not relevant for solving the task at hand.\n\n\n\n\n\n\nObservationSqueezeFilter\n - Removes redundant axes from the observation, which are axes with a dimension of 1.\n\n\n\n\n\n\nObservationStackingFilter\n - Stacks several observations on top of each other. For image observation this will\n create a 3D blob. The stacking is done in a lazy manner in order to reduce memory consumption. To achieve this,\n a LazyStack object is used in order to wrap the observations in the stack. For this reason, the\n ObservationStackingFilter \nmust\n be the last filter in the inputs filters stack.\n\n\n\n\n\n\nObservationUint8Filter\n - Converts a floating point observation into an unsigned int 8 bit observation. This is\n mostly useful for reducing memory consumption and is usually used for image observations. The filter will first\n spread the observation values over the range 0-255 and then discretize them into integer values.\n\n\n\n\n\n\nReward Filters\n\n\n\n\n\n\nRewardClippingFilter\n - Clips the reward values into a given range. For example, in DQN, the Atari rewards are\n clipped into the range -1 and 1 in order to control the scale of the returns.\n\n\n\n\n\n\nRewardNormalizationFilter\n - Normalizes the reward values with a running mean and standard deviation of\n all the rewards seen so far. When working with multiple workers, the statistics used for the normalization operation\n are accumulated over all the workers.\n\n\n\n\n\n\nRewardRescaleFilter\n - Rescales the reward by a given factor. Rescaling the rewards of the environment has been\n observed to have a large effect (negative or positive) on the behavior of the learning process.\nLine truncated
"title": "Filters"
},
{
"location": "/design/filters/#filters",
"text": "Filters are a mechanism in Coach that allows doing pre-processing and post-processing of the internal agent information.\nThere are two filter categories - Input filters - these are filters that process the information passed into the agent from the environment.\n This information includes the observation and the reward. Input filters therefore allow rescaling observations,\n normalizing rewards, stack observations, etc. Output filters - these are filters that process the information going out of the agent into the environment.\n This information includes the action the agent chooses to take. Output filters therefore allow conversion of\n actions from one space into another. For example, the agent can take N discrete actions, that will be mapped by\n the output filter onto N continuous actions. Filters can be stacked on top of each other in order to build complex processing flows of the inputs or outputs.",
"title": "Filters"
},
{
"location": "/design/filters/#input-filters",
"text": "The input filters are separated into two categories - observation filters and reward filters .",
"title": "Input Filters"
},
{
"location": "/design/filters/#observation-filters",
"text": "ObservationClippingFilter - Clips the observation values to a given range of values. For example, if the\n observation consists of measurements in an arbitrary range, and we want to control the minimum and maximum values\n of these observations, we can define a range and clip the values of the measurements. ObservationCropFilter - Crops the size of the observation to a given crop window. For example, in Atari, the\n observations are images with a shape of 210x160. Usually, we will want to crop the size of the observation to a\n square of 160x160 before rescaling them. ObservationMoveAxisFilter - Reorders the axes of the observation. This can be useful when the observation is an\n image, and we want to move the channel axis to be the last axis instead of the first axis. ObservationNormalizationFilter - Normalizes the observation values with a running mean and standard deviation of\n all the observations seen so far. The normalization is performed element-wise. Additionally, when working with\n multiple workers, the statistics used for the normalization operation are accumulated over all the workers. ObservationReductionBySubPartsNameFilter - Allows keeping only parts of the observation, by specifying their\n name. For example, the CARLA environment extracts multiple measurements that can be used by the agent, such as\n speed and location. If we want to only use the speed, it can be done using this filter. ObservationRescaleSizeByFactorFilter - Rescales an image observation by some factor. For example, the image size\n can be reduced by a factor of 2. ObservationRescaleToSizeFilter - Rescales an image observation to a given size. The target size does not\n necessarily keep the aspect ratio of the original observation. ObservationRGBToYFilter - Converts a color image observation specified using the RGB encoding into a grayscale\n image observation, by keeping only the luminance (Y) channel of the YUV encoding. This can be useful if the colors\n in the original image are not relevant for solving the task at hand. ObservationSqueezeFilter - Removes redundant axes from the observation, which are axes with a dimension of 1. ObservationStackingFilter - Stacks several observations on top of each other. For image observation this will\n create a 3D blob. The stacking is done in a lazy manner in order to reduce memory consumption. To achieve this,\n a LazyStack object is used in order to wrap the observations in the stack. For this reason, the\n ObservationStackingFilter must be the last filter in the inputs filters stack. ObservationUint8Filter - Converts a floating point observation into an unsigned int 8 bit observation. This is\n mostly useful for reducing memory consumption and is usually used for image observations. The filter will first\n spread the observation values over the range 0-255 and then discretize them into integer values.",
"title": "Observation Filters"
},
{
"location": "/design/filters/#reward-filters",
"text": "RewardClippingFilter - Clips the reward values into a given range. For example, in DQN, the Atari rewards are\n clipped into the range -1 and 1 in order to control the scale of the returns. RewardNormalizationFilter - Normalizes the reward values with a running mean and standard deviation of\n all the rewards seen so far. When working with multiple workers, the statistics used for the normalization operation\n are accumulated over all the workers. RewardRescaleFilter - Rescales the reward by a given factor. Rescaling the rewards of the environment has been\n observed to have a large effect (negative or positive) on the behavior of the learning process.",
"title": "Reward Filters"
},
{
"location": "/design/filters/#output-filters",
"text": "The output filters only process the actions.",
"title": "Output Filters"
},
{
"location": "/design/filters/#action-filters",
"text": "AttentionDiscretization - Discretizes an AttentionActionSpace . The attention action space defines the actions\n as choosing sub-boxes in a given box. For example, consider an image of size 100x100, where the action is choosing\n a crop window of size 20x20 to attend to in the image. AttentionDiscretization allows discretizing the possible crop\n windows to choose into a finite number of options, and map a discrete action space into those crop windows. BoxDiscretization - Discretizes a continuous action space into a discrete action space, allowing the usage of\n agents such as DQN for continuous environments such as MuJoCo. Given the number of bins to discretize into, the\n original continuous action space is uniformly separated into the given number of bins, each mapped to a discrete\n action index. For example, if the original actions space is between -1 and 1 and 5 bins were selected, the new action\n space will consist of 5 actions mapped to -1, -0.5, 0, 0.5 and 1. BoxMasking - Masks part of the action space to enforce the agent to work in a defined space. For example,\n if the original action space is between -1 and 1, then this filter can be used in order to constrain the agent actions\n to the range 0 and 1 instead. This essentially masks the range -1 and 0 from the agent. PartialDiscreteActionSpaceMap - Partial map of two countable action spaces. For example, consider an environment\n with a MultiSelect action space (select multiple actions at the same time, such as jump and go right), with 8 actual\n MultiSelect actions. If we want the agent to be able to select only 5 of those actions by their index (0-4), we can\n map a discrete action space with 5 actions into the 5 selected MultiSelect actions. This will both allow the agent to\n use regular discrete actions, and mask 3 of the actions from the agent. FullDiscreteActionSpaceMap - Full map of two countable action spaces. This works in a similar way to the\n PartialDiscreteActionSpaceMap, but maps the entire source action space into the entire target action space, without\n masking any actions. LinearBoxToBoxMap - A linear mapping of two box action spaces. For example, if the action space of the\n environment consists of continuous actions between 0 and 1, and we want the agent to choose actions between -1 and 1,\n the LinearBoxToBoxMap can be used to map the range -1 and 1 to the range 0 and 1 in a linear way. This means that the\n action -1 will be mapped to 0, the action 1 will be mapped to 1, and the rest of the actions will be linearly mapped\n between those values.",
"title": "Action Filters"
},
{
"location": "/algorithms/value_optimization/dqn/",
"text": "Deep Q Networks\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nPlaying Atari with Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\nUsing the next states from the sampled batch, run the target network to calculate the \n Q \n values for each of the actions \n Q(s_{t+1},a) \n, and keep only the maximum value for each state. \n\n\nIn order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. \n\n\n\n\nFor each action that was played, use the following equation for calculating the targets of the network:\u200b \n y_t=r(s_t,a_t)+\u03b3\\cdot max_a {Q(s_{t+1},a)} \n\n\n\n\n\n\n\n\nFinally, train the online network using the current states as inputs, and with the aforementioned targets. \n\n\n\n\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "DQN"
},
{
"location": "/algorithms/value_optimization/dqn/#deep-q-networks",
"text": "Actions space: Discrete References: Playing Atari with Deep Reinforcement Learning",
"title": "Deep Q Networks"
},
{
"location": "/algorithms/value_optimization/dqn/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/dqn/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/dqn/#training-the-network",
"text": "Sample a batch of transitions from the replay buffer. Using the next states from the sampled batch, run the target network to calculate the Q values for each of the actions Q(s_{t+1},a) , and keep only the maximum value for each state. In order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. For each action that was played, use the following equation for calculating the targets of the network:\u200b y_t=r(s_t,a_t)+\u03b3\\cdot max_a {Q(s_{t+1},a)} Finally, train the online network using the current states as inputs, and with the aforementioned targets. Once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/double_dqn/",
"text": "Double DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nDeep Reinforcement Learning with Double Q-learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\nUsing the next states from the sampled batch, run the online network in order to find the \nQ\n maximizing action \nargmax_a Q(s_{t+1},a)\n. For these actions, use the corresponding next states and run the target network to calculate \nQ(s_{t+1},argmax_a Q(s_{t+1},a))\n.\n\n\nIn order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. \n\n\n\n\nFor each action that was played, use the following equation for calculating the targets of the network:\n \n y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},argmax_a Q(s_{t+1},a)) \n\n\n\n\n\n\n\n\nFinally, train the online network using the current states as inputs, and with the aforementioned targets. \n\n\n\n\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Double DQN"
},
{
"location": "/algorithms/value_optimization/double_dqn/#double-dqn",
"text": "Actions space: Discrete References: Deep Reinforcement Learning with Double Q-learning",
"title": "Double DQN"
},
{
"location": "/algorithms/value_optimization/double_dqn/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/double_dqn/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/double_dqn/#training-the-network",
"text": "Sample a batch of transitions from the replay buffer. Using the next states from the sampled batch, run the online network in order to find the Q maximizing action argmax_a Q(s_{t+1},a) . For these actions, use the corresponding next states and run the target network to calculate Q(s_{t+1},argmax_a Q(s_{t+1},a)) . In order to zero out the updates for the actions that were not played (resulting from zeroing the MSE loss), use the current states from the sampled batch, and run the online network to get the current Q values predictions. Set those values as the targets for the actions that were not actually played. For each action that was played, use the following equation for calculating the targets of the network:\n y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},argmax_a Q(s_{t+1},a)) Finally, train the online network using the current states as inputs, and with the aforementioned targets. Once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/",
"text": "Dueling DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nDueling Network Architectures for Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nGeneral Description\n\n\nDueling DQN presents a change in the network structure comparing to DQN.\n\n\nDueling DQN uses a specialized \nDueling Q Head\n in order to separate \n Q \n to an \n A \n (advantage) stream and a \n V \n stream. Adding this type of structure to the network head allows the network to better differentiate actions from one another, and significantly improves the learning.\n\n\nIn many states, the values of the different actions are very similar, and it is less important which action to take.\nThis is especially important in environments where there are many actions to choose from. In DQN, on each training iteration, for each of the states in the batch, we update the \nQ\n values only for the specific actions taken in those states. This results in slower learning as we do not learn the \nQ\n values for actions that were not taken yet. On dueling architecture, on the other hand, learning is faster - as we start learning the state-value even if only a single action has been taken at this state.",
"title": "Dueling DQN"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/#dueling-dqn",
"text": "Actions space: Discrete References: Dueling Network Architectures for Deep Reinforcement Learning",
"title": "Dueling DQN"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/dueling_dqn/#general-description",
"text": "Dueling DQN presents a change in the network structure comparing to DQN. Dueling DQN uses a specialized Dueling Q Head in order to separate Q to an A (advantage) stream and a V stream. Adding this type of structure to the network head allows the network to better differentiate actions from one another, and significantly improves the learning. In many states, the values of the different actions are very similar, and it is less important which action to take.\nThis is especially important in environments where there are many actions to choose from. In DQN, on each training iteration, for each of the states in the batch, we update the Q values only for the specific actions taken in those states. This results in slower learning as we do not learn the Q values for actions that were not taken yet. On dueling architecture, on the other hand, learning is faster - as we start learning the state-value even if only a single action has been taken at this state.",
"title": "General Description"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/",
"text": "Categorical DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nA Distributional Perspective on Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\n\n\nThe Bellman update is projected to the set of atoms representing the \n Q \n values distribution, such that the \ni-th\n component of the projected update is calculated as follows:\n \n (\\Phi \\hat{T} Z_{\\theta}(s_t,a_t))_i=\\sum_{j=0}^{N-1}\\Big[1-\\frac{|[\\hat{T}_{z_{j}}]^{V_{MAX}}_{V_{MIN}}-z_i|}{\\Delta z}\\Big]^1_0 \\ p_j(s_{t+1}, \\pi(s_{t+1})) \n\n where:\n\n\n\n\n\n\n[ \\cdot ] \n bounds its argument in the range [a, b]\n\n\n\n\n\\hat{T}_{z_{j}}\n is the Bellman update for atom \nz_j\n: \u00a0 \u00a0 \n\\hat{T}_{z_{j}} := r+\\gamma z_j\n\n\n\n\n\n\n\n\n\n\nNetwork is trained with the cross entropy loss between the resulting probability distribution and the target probability distribution. Only the target of the actions that were actually taken is updated. \n\n\n\n\nOnce in every few thousand steps, weights are copied from the online network to the target network.",
"title": "Categorical DQN"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/#categorical-dqn",
"text": "Actions space: Discrete References: A Distributional Perspective on Reinforcement Learning",
"title": "Categorical DQN"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/categorical_dqn/#training-the-network",
"text": "Sample a batch of transitions from the replay buffer. The Bellman update is projected to the set of atoms representing the Q values distribution, such that the i-th component of the projected update is calculated as follows:\n (\\Phi \\hat{T} Z_{\\theta}(s_t,a_t))_i=\\sum_{j=0}^{N-1}\\Big[1-\\frac{|[\\hat{T}_{z_{j}}]^{V_{MAX}}_{V_{MIN}}-z_i|}{\\Delta z}\\Big]^1_0 \\ p_j(s_{t+1}, \\pi(s_{t+1})) \n where: [ \\cdot ] bounds its argument in the range [a, b] \\hat{T}_{z_{j}} is the Bellman update for atom z_j : \u00a0 \u00a0 \\hat{T}_{z_{j}} := r+\\gamma z_j Network is trained with the cross entropy loss between the resulting probability distribution and the target probability distribution. Only the target of the actions that were actually taken is updated. Once in every few thousand steps, weights are copied from the online network to the target network.",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/mmc/",
"text": "Mixed Monte Carlo\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nCount-Based Exploration with Neural Density Models\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\nIn MMC, targets are calculated as a mixture between Double DQN targets and full Monte Carlo samples (total discounted returns).\n\n\nThe DDQN targets are calculated in the same manner as in the DDQN agent:\n\n\n\n\n y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) \n\n\n\n\nThe Monte Carlo targets are calculated by summing up the discounted rewards across the entire episode:\n\n\n\n\n y_t^{MC}=\\sum_{j=0}^T\\gamma^j r(s_{t+j},a_{t+j} ) \n\n\n\n\nA mixing ratio \n\\alpha\n is then used to get the final targets:\n\n\n\n\n y_t=(1-\\alpha)\\cdot y_t^{DDQN}+\\alpha \\cdot y_t^{MC} \n\n\n\n\nFinally, the online network is trained using the current states as inputs, and the calculated targets.\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Mixed Monte Carlo"
},
{
"location": "/algorithms/value_optimization/mmc/#mixed-monte-carlo",
"text": "Actions space: Discrete References: Count-Based Exploration with Neural Density Models",
"title": "Mixed Monte Carlo"
},
{
"location": "/algorithms/value_optimization/mmc/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/mmc/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/mmc/#training-the-network",
"text": "In MMC, targets are calculated as a mixture between Double DQN targets and full Monte Carlo samples (total discounted returns). The DDQN targets are calculated in the same manner as in the DDQN agent: y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) The Monte Carlo targets are calculated by summing up the discounted rewards across the entire episode: y_t^{MC}=\\sum_{j=0}^T\\gamma^j r(s_{t+j},a_{t+j} ) A mixing ratio \\alpha is then used to get the final targets: y_t=(1-\\alpha)\\cdot y_t^{DDQN}+\\alpha \\cdot y_t^{MC} Finally, the online network is trained using the current states as inputs, and the calculated targets.\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/pal/",
"text": "Persistent Advantage Learning\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nIncreasing the Action Gap: New Operators for Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\n\n\n\n\nSample a batch of transitions from the replay buffer. \n\n\n\n\n\n\nStart by calculating the initial target values in the same manner as they are calculated in DDQN\n \n y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) \n\n\n\n\n\n\nThe action gap \n V(s_t )-Q(s_t,a_t) \n should then be subtracted from each of the calculated targets. To calculate the action gap, run the target network using the current states and get the \n Q \n values for all the actions. Then estimate \n V \n as the maximum predicted \n Q \n value for the current state:\n \n V(s_t )=max_a Q(s_t,a) \n\n\n\n\nFor \nadvantage learning (AL)\n, reduce the action gap weighted by a predefined parameter \n \\alpha \n from the targets \n y_t^{DDQN} \n: \n \n y_t=y_t^{DDQN}-\\alpha \\cdot (V(s_t )-Q(s_t,a_t )) \n\n\n\n\nFor \npersistent advantage learning (PAL)\n, the target network is also used in order to calculate the action gap for the next state:\n \n V(s_{t+1} )-Q(s_{t+1},a_{t+1}) \n\n where \n a_{t+1} \n is chosen by running the next states through the online network and choosing the action that has the highest predicted \n Q \n value. Finally, the targets will be defined as -\n \n y_t=y_t^{DDQN}-\\alpha \\cdot min(V(s_t )-Q(s_t,a_t ),V(s_{t+1} )-Q(s_{t+1},a_{t+1} )) \n\n\n\n\n\n\nTrain the online network using the current states as inputs, and with the aforementioned targets.\n\n\n\n\n\n\nOnce in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Persistent Advantage Learning"
},
{
"location": "/algorithms/value_optimization/pal/#persistent-advantage-learning",
"text": "Actions space: Discrete References: Increasing the Action Gap: New Operators for Reinforcement Learning",
"title": "Persistent Advantage Learning"
},
{
"location": "/algorithms/value_optimization/pal/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/pal/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/pal/#training-the-network",
"text": "Sample a batch of transitions from the replay buffer. Start by calculating the initial target values in the same manner as they are calculated in DDQN\n y_t^{DDQN}=r(s_t,a_t )+\\gamma Q(s_{t+1},argmax_a Q(s_{t+1},a)) The action gap V(s_t )-Q(s_t,a_t) should then be subtracted from each of the calculated targets. To calculate the action gap, run the target network using the current states and get the Q values for all the actions. Then estimate V as the maximum predicted Q value for the current state:\n V(s_t )=max_a Q(s_t,a) For advantage learning (AL) , reduce the action gap weighted by a predefined parameter \\alpha from the targets y_t^{DDQN} : \n y_t=y_t^{DDQN}-\\alpha \\cdot (V(s_t )-Q(s_t,a_t )) For persistent advantage learning (PAL) , the target network is also used in order to calculate the action gap for the next state:\n V(s_{t+1} )-Q(s_{t+1},a_{t+1}) \n where a_{t+1} is chosen by running the next states through the online network and choosing the action that has the highest predicted Q value. Finally, the targets will be defined as -\n y_t=y_t^{DDQN}-\\alpha \\cdot min(V(s_t )-Q(s_t,a_t ),V(s_{t+1} )-Q(s_{t+1},a_{t+1} )) Train the online network using the current states as inputs, and with the aforementioned targets. Once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/nec/",
"text": "Neural Episodic Control\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nNeural Episodic Control\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\n\n\nUse the current state as an input to the online network and extract the state embedding, which is the intermediate output from the middleware. \n\n\nFor each possible action \na_i\n, run the DND head using the state embedding and the selected action \na_i\n as inputs. The DND is queried and returns the \n P \n nearest neighbor keys and values. The keys and values are used to calculate and return the action \n Q \n value from the network. \n\n\nPass all the \n Q \n values to the exploration policy and choose an action accordingly. \n\n\nStore the state embeddings and actions taken during the current episode in a small buffer \nB\n, in order to accumulate transitions until it is possible to calculate the total discounted returns over the entire episode.\n\n\n\n\nFinalizing an episode\n\n\nFor each step in the episode, the state embeddings and the taken actions are stored in the buffer \nB\n. When the episode is finished, the replay buffer calculates the \n N \n-step total return of each transition in the buffer, bootstrapped using the maximum \nQ\n value of the \nN\n-th transition. Those values are inserted along with the total return into the DND, and the buffer \nB\n is reset.\n\n\nTraining the network\n\n\nTrain the network only when the DND has enough entries for querying.\n\n\nTo train the network, the current states are used as the inputs and the \nN\n-step returns are used as the targets. The \nN\n-step return used takes into account \n N \n consecutive steps, and bootstraps the last value from the network if necessary:\n\n y_t=\\sum_{j=0}^{N-1}\\gamma^j r(s_{t+j},a_{t+j} ) +\\gamma^N max_a Q(s_{t+N},a)",
"title": "Neural Episodic Control"
},
{
"location": "/algorithms/value_optimization/nec/#neural-episodic-control",
"text": "Actions space: Discrete References: Neural Episodic Control",
"title": "Neural Episodic Control"
},
{
"location": "/algorithms/value_optimization/nec/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/nec/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/nec/#choosing-an-action",
"text": "Use the current state as an input to the online network and extract the state embedding, which is the intermediate output from the middleware. For each possible action a_i , run the DND head using the state embedding and the selected action a_i as inputs. The DND is queried and returns the P nearest neighbor keys and values. The keys and values are used to calculate and return the action Q value from the network. Pass all the Q values to the exploration policy and choose an action accordingly. Store the state embeddings and actions taken during the current episode in a small buffer B , in order to accumulate transitions until it is possible to calculate the total discounted returns over the entire episode.",
"title": "Choosing an action"
},
{
"location": "/algorithms/value_optimization/nec/#finalizing-an-episode",
"text": "For each step in the episode, the state embeddings and the taken actions are stored in the buffer B . When the episode is finished, the replay buffer calculates the N -step total return of each transition in the buffer, bootstrapped using the maximum Q value of the N -th transition. Those values are inserted along with the total return into the DND, and the buffer B is reset.",
"title": "Finalizing an episode"
},
{
"location": "/algorithms/value_optimization/nec/#training-the-network",
"text": "Train the network only when the DND has enough entries for querying. To train the network, the current states are used as the inputs and the N -step returns are used as the targets. The N -step return used takes into account N consecutive steps, and bootstraps the last value from the network if necessary: y_t=\\sum_{j=0}^{N-1}\\gamma^j r(s_{t+j},a_{t+j} ) +\\gamma^N max_a Q(s_{t+N},a)",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/bs_dqn/",
"text": "Bootstrapped DQN\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nDeep Exploration via Bootstrapped DQN\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\nThe current states are used as the input to the network. The network contains several \nQ\n heads, which are used for returning different estimations of the action \n Q \n values. For each episode, the bootstrapped exploration policy selects a single head to play with during the episode. According to the selected head, only the relevant output \n Q \n values are used. Using those \n Q \n values, the exploration policy then selects the action for acting.\n\n\nStoring the transitions\n\n\nFor each transition, a Binomial mask is generated according to a predefined probability, and the number of output heads. The mask is a binary vector where each element holds a 0 for heads that shouldn't train on the specific transition, and 1 for heads that should use the transition for training. The mask is stored as part of the transition info in the replay buffer. \n\n\nTraining the network\n\n\nFirst, sample a batch of transitions from the replay buffer. Run the current states through the network and get the current \n Q \n value predictions for all the heads and all the actions. For each transition in the batch, and for each output head, if the transition mask is 1 - change the targets of the played action to \ny_t\n, according to the standard DQN update rule:\n\n\n\n\n y_t=r(s_t,a_t )+\\gamma\\cdot max_a Q(s_{t+1},a) \n\n\n\n\nOtherwise, leave it intact so that the transition does not affect the learning of this head. Then, train the online network according to the calculated targets.\n\n\nAs in DQN, once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Bootstrapped DQN"
},
{
"location": "/algorithms/value_optimization/bs_dqn/#bootstrapped-dqn",
"text": "Actions space: Discrete References: Deep Exploration via Bootstrapped DQN",
"title": "Bootstrapped DQN"
},
{
"location": "/algorithms/value_optimization/bs_dqn/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/bs_dqn/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/bs_dqn/#choosing-an-action",
"text": "The current states are used as the input to the network. The network contains several Q heads, which are used for returning different estimations of the action Q values. For each episode, the bootstrapped exploration policy selects a single head to play with during the episode. According to the selected head, only the relevant output Q values are used. Using those Q values, the exploration policy then selects the action for acting.",
"title": "Choosing an action"
},
{
"location": "/algorithms/value_optimization/bs_dqn/#storing-the-transitions",
"text": "For each transition, a Binomial mask is generated according to a predefined probability, and the number of output heads. The mask is a binary vector where each element holds a 0 for heads that shouldn't train on the specific transition, and 1 for heads that should use the transition for training. The mask is stored as part of the transition info in the replay buffer.",
"title": "Storing the transitions"
},
{
"location": "/algorithms/value_optimization/bs_dqn/#training-the-network",
"text": "First, sample a batch of transitions from the replay buffer. Run the current states through the network and get the current Q value predictions for all the heads and all the actions. For each transition in the batch, and for each output head, if the transition mask is 1 - change the targets of the played action to y_t , according to the standard DQN update rule: y_t=r(s_t,a_t )+\\gamma\\cdot max_a Q(s_{t+1},a) Otherwise, leave it intact so that the transition does not affect the learning of this head. Then, train the online network according to the calculated targets. As in DQN, once in every few thousand steps, copy the weights from the online network to the target network.",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/n_step/",
"text": "N-Step Q Learning\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nAsynchronous Methods for Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\nThe \nN\n-step Q learning algorithm works in similar manner to DQN except for the following changes:\n\n\n\n\n\n\nNo replay buffer is used. Instead of sampling random batches of transitions, the network is trained every \nN\n steps using the latest \nN\n steps played by the agent.\n\n\n\n\n\n\nIn order to stabilize the learning, multiple workers work together to update the network. This creates the same effect as uncorrelating the samples used for training.\n\n\n\n\n\n\nInstead of using single-step Q targets for the network, the rewards from \nN\n consequent steps are accumulated to form the \nN\n-step Q targets, according to the following equation: \n\nR(s_t, a_t) = \\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k})\n\nwhere \nk\n is \nT_{max} - State\\_Index\n for each state in the batch",
"title": "N-Step Q Learning"
},
{
"location": "/algorithms/value_optimization/n_step/#n-step-q-learning",
"text": "Actions space: Discrete References: Asynchronous Methods for Deep Reinforcement Learning",
"title": "N-Step Q Learning"
},
{
"location": "/algorithms/value_optimization/n_step/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/n_step/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/n_step/#training-the-network",
"text": "The N -step Q learning algorithm works in similar manner to DQN except for the following changes: No replay buffer is used. Instead of sampling random batches of transitions, the network is trained every N steps using the latest N steps played by the agent. In order to stabilize the learning, multiple workers work together to update the network. This creates the same effect as uncorrelating the samples used for training. Instead of using single-step Q targets for the network, the rewards from N consequent steps are accumulated to form the N -step Q targets, according to the following equation: R(s_t, a_t) = \\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k}) \nwhere k is T_{max} - State\\_Index for each state in the batch",
"title": "Training the network"
},
{
"location": "/algorithms/value_optimization/naf/",
"text": "Normalized Advantage Functions\n\n\nActions space:\n Continuous\n\n\nReferences:\n \nContinuous Deep Q-Learning with Model-based Acceleration\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\nThe current state is used as an input to the network. The action mean \n \\mu(s_t ) \n is extracted from the output head. It is then passed to the exploration policy which adds noise in order to encourage exploration.\n\n\nTraining the network\n\n\nThe network is trained by using the following targets:\n\n y_t=r(s_t,a_t )+\\gamma\\cdot V(s_{t+1}) \n\nUse the next states as the inputs to the target network and extract the \n V \n value, from within the head, to get \n V(s_{t+1} ) \n. Then, update the online network using the current states and actions as inputs, and \n y_t \n as the targets.\nAfter every training step, use a soft update in order to copy the weights from the online network to the target network.",
"title": "Normalized Advantage Functions"
},
{
"location": "/algorithms/value_optimization/naf/#normalized-advantage-functions",
"text": "Actions space: Continuous References: Continuous Deep Q-Learning with Model-based Acceleration",
"title": "Normalized Advantage Functions"
},
{
"location": "/algorithms/value_optimization/naf/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/value_optimization/naf/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/value_optimization/naf/#choosing-an-action",
"text": "The current state is used as an input to the network. The action mean \\mu(s_t ) is extracted from the output head. It is then passed to the exploration policy which adds noise in order to encourage exploration.",
"title": "Choosing an action"
},
{
"location": "/algorithms/value_optimization/naf/#training-the-network",
"text": "The network is trained by using the following targets: y_t=r(s_t,a_t )+\\gamma\\cdot V(s_{t+1}) \nUse the next states as the inputs to the target network and extract the V value, from within the head, to get V(s_{t+1} ) . Then, update the online network using the current states and actions as inputs, and y_t as the targets.\nAfter every training step, use a soft update in order to copy the weights from the online network to the target network.",
"title": "Training the network"
},
{
"location": "/algorithms/policy_optimization/pg/",
"text": "Policy Gradient\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nSimple Statistical Gradient-Following Algorithms for Connectionist Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Discrete actions\n\n\nRun the current states through the network and get a policy distribution over the actions. While training, sample from the policy distribution. When testing, take the action with the highest probability. \n\n\nTraining the network\n\n\nThe policy head loss is defined as \n L=-log (\\pi) \\cdot PolicyGradientRescaler \n. The \nPolicyGradientRescaler\n is used in order to reduce the policy gradient variance, which might be very noisy. This is done in order to reduce the variance of the updates, since noisy gradient updates might destabilize the policy's convergence. The rescaler is a configurable parameter and there are few options to choose from: \n\n\n \nTotal Episode Return\n - The sum of all the discounted rewards during the episode.\n\n \nFuture Return\n - Return from each transition until the end of the episode.\n\n \nFuture Return Normalized by Episode\n - Future returns across the episode normalized by the episode's mean and standard deviation.\n\n \nFuture Return Normalized by Timestep\n - Future returns normalized using running means and standard deviations, which are calculated seperately for each timestep, across different episodes. \n\n\nGradients are accumulated over a number of full played episodes. The gradients accumulation over several episodes serves the same purpose - reducing the update variance. After accumulating gradients for several episodes, the gradients are then applied to the network.",
"title": "Policy Gradient"
},
{
"location": "/algorithms/policy_optimization/pg/#policy-gradient",
"text": "Actions space: Discrete|Continuous References: Simple Statistical Gradient-Following Algorithms for Connectionist Reinforcement Learning",
"title": "Policy Gradient"
},
{
"location": "/algorithms/policy_optimization/pg/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/pg/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/pg/#choosing-an-action-discrete-actions",
"text": "Run the current states through the network and get a policy distribution over the actions. While training, sample from the policy distribution. When testing, take the action with the highest probability.",
"title": "Choosing an action - Discrete actions"
},
{
"location": "/algorithms/policy_optimization/pg/#training-the-network",
"text": "The policy head loss is defined as L=-log (\\pi) \\cdot PolicyGradientRescaler . The PolicyGradientRescaler is used in order to reduce the policy gradient variance, which might be very noisy. This is done in order to reduce the variance of the updates, since noisy gradient updates might destabilize the policy's convergence. The rescaler is a configurable parameter and there are few options to choose from: Total Episode Return - The sum of all the discounted rewards during the episode. Future Return - Return from each transition until the end of the episode. Future Return Normalized by Episode - Future returns across the episode normalized by the episode's mean and standard deviation. Future Return Normalized by Timestep - Future returns normalized using running means and standard deviations, which are calculated seperately for each timestep, across different episodes. Gradients are accumulated over a number of full played episodes. The gradients accumulation over several episodes serves the same purpose - reducing the update variance. After accumulating gradients for several episodes, the gradients are then applied to the network.",
"title": "Training the network"
},
{
"location": "/algorithms/policy_optimization/ac/",
"text": "Actor-Critic\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nAsynchronous Methods for Deep Reinforcement Learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Discrete actions\n\n\nThe policy network is used in order to predict action probabilites. While training, a sample is taken from a categorical distribution assigned with these probabilities. When testing, the action with the highest probability is used.\n\n\nTraining the network\n\n\nA batch of \n T_{max} \n transitions is used, and the advantages are calculated upon it.\n\n\nAdvantages can be calculated by either of the following methods (configured by the selected preset) -\n\n\n\n\nA_VALUE\n - Estimating advantage directly:\n A(s_t, a_t) = \\underbrace{\\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k})}_{Q(s_t, a_t)} - V(s_t) \nwhere \nk\n is \nT_{max} - State\\_Index\n for each state in the batch.\n\n\nGAE\n - By following the \nGeneralized Advantage Estimation\n paper. \n\n\n\n\nThe advantages are then used in order to accumulate gradients according to \n\n L = -\\mathop{\\mathbb{E}} [log (\\pi) \\cdot A]",
"title": "Actor-Critic"
},
{
"location": "/algorithms/policy_optimization/ac/#actor-critic",
"text": "Actions space: Discrete|Continuous References: Asynchronous Methods for Deep Reinforcement Learning",
"title": "Actor-Critic"
},
{
"location": "/algorithms/policy_optimization/ac/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/ac/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/ac/#choosing-an-action-discrete-actions",
"text": "The policy network is used in order to predict action probabilites. While training, a sample is taken from a categorical distribution assigned with these probabilities. When testing, the action with the highest probability is used.",
"title": "Choosing an action - Discrete actions"
},
{
"location": "/algorithms/policy_optimization/ac/#training-the-network",
"text": "A batch of T_{max} transitions is used, and the advantages are calculated upon it. Advantages can be calculated by either of the following methods (configured by the selected preset) - A_VALUE - Estimating advantage directly: A(s_t, a_t) = \\underbrace{\\sum_{i=t}^{i=t + k - 1} \\gamma^{i-t}r_i +\\gamma^{k} V(s_{t+k})}_{Q(s_t, a_t)} - V(s_t) where k is T_{max} - State\\_Index for each state in the batch. GAE - By following the Generalized Advantage Estimation paper. The advantages are then used in order to accumulate gradients according to L = -\\mathop{\\mathbb{E}} [log (\\pi) \\cdot A]",
"title": "Training the network"
},
{
"location": "/algorithms/policy_optimization/ddpg/",
"text": "Deep Deterministic Policy Gradient\n\n\nActions space:\n Continuous\n\n\nReferences:\n \nContinuous control with deep reinforcement learning\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\nPass the current states through the actor network, and get an action mean vector \n \\mu \n. While in training phase, use a continuous exploration policy, such as the Ornstein-Uhlenbeck process, to add exploration noise to the action. When testing, use the mean vector \n\\mu\n as-is.\n\n\nTraining the network\n\n\nStart by sampling a batch of transitions from the experience replay.\n\n\n\n\nTo train the \ncritic network\n, use the following targets:\n\n\n\n\n\n\n y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},\\mu(s_{t+1} )) \n\n First run the actor target network, using the next states as the inputs, and get \n \\mu (s_{t+1} ) \n. Next, run the critic target network using the next states and \n \\mu (s_{t+1} ) \n, and use the output to calculate \n y_t \n according to the equation above. To train the network, use the current states and actions as the inputs, and \ny_t\n as the targets.\n\n\n\n\nTo train the \nactor network\n, use the following equation:\n\n\n\n\n\n\n \\nabla_{\\theta^\\mu } J \\approx E_{s_t \\tilde{} \\rho^\\beta } [\\nabla_a Q(s,a)|_{s=s_t,a=\\mu (s_t ) } \\cdot \\nabla_{\\theta^\\mu} \\mu(s)|_{s=s_t} ] \n\n Use the actor's online network to get the action mean values using the current states as the inputs. Then, use the critic online network in order to get the gradients of the critic output with respect to the action mean values \n \\nabla _a Q(s,a)|_{s=s_t,a=\\mu(s_t ) } \n. Using the chain rule, calculate the gradients of the actor's output, with respect to the actor weights, given \n \\nabla_a Q(s,a) \n. Finally, apply those gradients to the actor network.\n\n\nAfter every training step, do a soft update of the critic and actor target networks' weights from the online networks.",
"title": "Deep Determinstic Policy Gradients"
},
{
"location": "/algorithms/policy_optimization/ddpg/#deep-deterministic-policy-gradient",
"text": "Actions space: Continuous References: Continuous control with deep reinforcement learning",
"title": "Deep Deterministic Policy Gradient"
},
{
"location": "/algorithms/policy_optimization/ddpg/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/ddpg/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/ddpg/#choosing-an-action",
"text": "Pass the current states through the actor network, and get an action mean vector \\mu . While in training phase, use a continuous exploration policy, such as the Ornstein-Uhlenbeck process, to add exploration noise to the action. When testing, use the mean vector \\mu as-is.",
"title": "Choosing an action"
},
{
"location": "/algorithms/policy_optimization/ddpg/#training-the-network",
"text": "Start by sampling a batch of transitions from the experience replay. To train the critic network , use the following targets: y_t=r(s_t,a_t )+\\gamma \\cdot Q(s_{t+1},\\mu(s_{t+1} )) \n First run the actor target network, using the next states as the inputs, and get \\mu (s_{t+1} ) . Next, run the critic target network using the next states and \\mu (s_{t+1} ) , and use the output to calculate y_t according to the equation above. To train the network, use the current states and actions as the inputs, and y_t as the targets. To train the actor network , use the following equation: \\nabla_{\\theta^\\mu } J \\approx E_{s_t \\tilde{} \\rho^\\beta } [\\nabla_a Q(s,a)|_{s=s_t,a=\\mu (s_t ) } \\cdot \\nabla_{\\theta^\\mu} \\mu(s)|_{s=s_t} ] \n Use the actor's online network to get the action mean values using the current states as the inputs. Then, use the critic online network in order to get the gradients of the critic output with respect to the action mean values \\nabla _a Q(s,a)|_{s=s_t,a=\\mu(s_t ) } . Using the chain rule, calculate the gradients of the actor's output, with respect to the actor weights, given \\nabla_a Q(s,a) . Finally, apply those gradients to the actor network. After every training step, do a soft update of the critic and actor target networks' weights from the online networks.",
"title": "Training the network"
},
{
"location": "/algorithms/policy_optimization/ppo/",
"text": "Proximal Policy Optimization\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nProximal Policy Optimization Algorithms\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Continuous actions\n\n\nRun the observation through the policy network, and get the mean and standard deviation vectors for this observation. While in training phase, sample from a multi-dimensional Gaussian distribution with these mean and standard deviation values. When testing, just take the mean values predicted by the network. \n\n\nTraining the network\n\n\n\n\nCollect a big chunk of experience (in the order of thousands of transitions, sampled from multiple episodes).\n\n\nCalculate the advantages for each transition, using the \nGeneralized Advantage Estimation\n method (Schulman '2015). \n\n\nRun a single training iteration of the value network using an L-BFGS optimizer. Unlike first order optimizers, the L-BFGS optimizer runs on the entire dataset at once, without batching. It continues running until some low loss threshold is reached. To prevent overfitting to the current dataset, the value targets are updated in a soft manner, using an Exponentially Weighted Moving Average, based on the total discounted returns of each state in each episode.\n\n\nRun several training iterations of the policy network. This is done by using the previously calculated advantages as targets. The loss function penalizes policies that deviate too far from the old policy (the policy that was used \nbefore\n starting to run the current set of training iterations) using a regularization term. \n\n\nAfter training is done, the last sampled KL divergence value will be compared with the \ntarget KL divergence\n value, in order to adapt the penalty coefficient used in the policy loss. If the KL divergence went too high, increase the penalty, if it went too low, reduce it. Otherwise, leave it unchanged.",
"title": "Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/ppo/#proximal-policy-optimization",
"text": "Actions space: Discrete|Continuous References: Proximal Policy Optimization Algorithms",
"title": "Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/ppo/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/ppo/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/ppo/#choosing-an-action-continuous-actions",
"text": "Run the observation through the policy network, and get the mean and standard deviation vectors for this observation. While in training phase, sample from a multi-dimensional Gaussian distribution with these mean and standard deviation values. When testing, just take the mean values predicted by the network.",
"title": "Choosing an action - Continuous actions"
},
{
"location": "/algorithms/policy_optimization/ppo/#training-the-network",
"text": "Collect a big chunk of experience (in the order of thousands of transitions, sampled from multiple episodes). Calculate the advantages for each transition, using the Generalized Advantage Estimation method (Schulman '2015). Run a single training iteration of the value network using an L-BFGS optimizer. Unlike first order optimizers, the L-BFGS optimizer runs on the entire dataset at once, without batching. It continues running until some low loss threshold is reached. To prevent overfitting to the current dataset, the value targets are updated in a soft manner, using an Exponentially Weighted Moving Average, based on the total discounted returns of each state in each episode. Run several training iterations of the policy network. This is done by using the previously calculated advantages as targets. The loss function penalizes policies that deviate too far from the old policy (the policy that was used before starting to run the current set of training iterations) using a regularization term. After training is done, the last sampled KL divergence value will be compared with the target KL divergence value, in order to adapt the penalty coefficient used in the policy loss. If the KL divergence went too high, increase the penalty, if it went too low, reduce it. Otherwise, leave it unchanged.",
"title": "Training the network"
},
{
"location": "/algorithms/policy_optimization/cppo/",
"text": "Clipped Proximal Policy Optimization\n\n\nActions space:\n Discrete|Continuous\n\n\nReferences:\n \nProximal Policy Optimization Algorithms\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action - Continuous action\n\n\nSame as in PPO. \n\n\nTraining the network\n\n\nVery similar to PPO, with several small (but very simplifying) changes:\n\n\n\n\n\n\nTrain both the value and policy networks, simultaneously, by defining a single loss function, which is the sum of each of the networks loss functions. Then, back propagate gradients only once from this unified loss function.\n\n\n\n\n\n\nThe unified network's optimizer is set to Adam (instead of L-BFGS for the value network as in PPO). \n\n\n\n\n\n\nValue targets are now also calculated based on the GAE advantages. In this method, the \n V \n values are predicted from the critic network, and then added to the GAE based advantages, in order to get a \n Q \n value for each action. Now, since our critic network is predicting a \n V \n value for each state, setting the \n Q \n calculated action-values as a target, will on average serve as a \n V \n state-value target. \n\n\n\n\n\n\nInstead of adapting the penalizing KL divergence coefficient used in PPO, the likelihood ratio \nr_t(\\theta) =\\frac{\\pi_{\\theta}(a|s)}{\\pi_{\\theta_{old}}(a|s)}\n is clipped, to achieve a similar effect. This is done by defining the policy's loss function to be the minimum between the standard surrogate loss and an epsilon clipped surrogate loss:\n\n\n\n\n\n\n\n\nL^{CLIP}(\\theta)=E_{t}[min(r_t(\\theta)\\cdot \\hat{A}_t, clip(r_t(\\theta), 1-\\epsilon, 1+\\epsilon) \\cdot \\hat{A}_t)]",
"title": "Clipped Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/cppo/#clipped-proximal-policy-optimization",
"text": "Actions space: Discrete|Continuous References: Proximal Policy Optimization Algorithms",
"title": "Clipped Proximal Policy Optimization"
},
{
"location": "/algorithms/policy_optimization/cppo/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/policy_optimization/cppo/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/policy_optimization/cppo/#choosing-an-action-continuous-action",
"text": "Same as in PPO.",
"title": "Choosing an action - Continuous action"
},
{
"location": "/algorithms/policy_optimization/cppo/#training-the-network",
"text": "Very similar to PPO, with several small (but very simplifying) changes: Train both the value and policy networks, simultaneously, by defining a single loss function, which is the sum of each of the networks loss functions. Then, back propagate gradients only once from this unified loss function. The unified network's optimizer is set to Adam (instead of L-BFGS for the value network as in PPO). Value targets are now also calculated based on the GAE advantages. In this method, the V values are predicted from the critic network, and then added to the GAE based advantages, in order to get a Q value for each action. Now, since our critic network is predicting a V value for each state, setting the Q calculated action-values as a target, will on average serve as a V state-value target. Instead of adapting the penalizing KL divergence coefficient used in PPO, the likelihood ratio r_t(\\theta) =\\frac{\\pi_{\\theta}(a|s)}{\\pi_{\\theta_{old}}(a|s)} is clipped, to achieve a similar effect. This is done by defining the policy's loss function to be the minimum between the standard surrogate loss and an epsilon clipped surrogate loss: L^{CLIP}(\\theta)=E_{t}[min(r_t(\\theta)\\cdot \\hat{A}_t, clip(r_t(\\theta), 1-\\epsilon, 1+\\epsilon) \\cdot \\hat{A}_t)]",
"title": "Training the network"
},
{
"location": "/algorithms/other/dfp/",
"text": "Direct Future Prediction\n\n\nActions space:\n Discrete\n\n\nReferences:\n \nLearning to Act by Predicting the Future\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nChoosing an action\n\n\n\n\nThe current states (observations and measurements) and the corresponding goal vector are passed as an input to the network. The output of the network is the predicted future measurements for time-steps \nt+1,t+2,t+4,t+8,t+16\n and \nt+32\n for each possible action. \n\n\nFor each action, the measurements of each predicted time-step are multiplied by the goal vector, and the result is a single vector of future values for each action. \n\n\nThen, a weighted sum of the future values of each action is calculated, and the result is a single value for each action. \n\n\nThe action values are passed to the exploration policy to decide on the action to use.\n\n\n\n\nTraining the network\n\n\nGiven a batch of transitions, run them through the network to get the current predictions of the future measurements per action, and set them as the initial targets for training the network. For each transition \n(s_t,a_t,r_t,s_{t+1} )\n in the batch, the target of the network for the action that was taken, is the actual measurements that were seen in time-steps \nt+1,t+2,t+4,t+8,t+16\n and \nt+32\n. For the actions that were not taken, the targets are the current values.",
"title": "Direct Future Prediction"
},
{
"location": "/algorithms/other/dfp/#direct-future-prediction",
"text": "Actions space: Discrete References: Learning to Act by Predicting the Future",
"title": "Direct Future Prediction"
},
{
"location": "/algorithms/other/dfp/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/other/dfp/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/other/dfp/#choosing-an-action",
"text": "The current states (observations and measurements) and the corresponding goal vector are passed as an input to the network. The output of the network is the predicted future measurements for time-steps t+1,t+2,t+4,t+8,t+16 and t+32 for each possible action. For each action, the measurements of each predicted time-step are multiplied by the goal vector, and the result is a single vector of future values for each action. Then, a weighted sum of the future values of each action is calculated, and the result is a single value for each action. The action values are passed to the exploration policy to decide on the action to use.",
"title": "Choosing an action"
},
{
"location": "/algorithms/other/dfp/#training-the-network",
"text": "Given a batch of transitions, run them through the network to get the current predictions of the future measurements per action, and set them as the initial targets for training the network. For each transition (s_t,a_t,r_t,s_{t+1} ) in the batch, the target of the network for the action that was taken, is the actual measurements that were seen in time-steps t+1,t+2,t+4,t+8,t+16 and t+32 . For the actions that were not taken, the targets are the current values.",
"title": "Training the network"
},
{
"location": "/algorithms/imitation/bc/",
"text": "Behavioral Cloning\n\n\nActions space:\n Discrete|Continuous\n\n\nNetwork Structure\n\n\n\n\n\n\n\n\n\n\n\nAlgorithm Description\n\n\nTraining the network\n\n\nThe replay buffer contains the expert demonstrations for the task.\nThese demonstrations are given as state, action tuples, and with no reward.\nThe training goal is to reduce the difference between the actions predicted by the network and the actions taken by the expert for each state.\n\n\n\n\nSample a batch of transitions from the replay buffer.\n\n\nUse the current states as input to the network, and the expert actions as the targets of the network.\n\n\nThe loss function for the network is MSE, and therefore we use the Q head to minimize this loss.",
"title": "Behavioral Cloning"
},
{
"location": "/algorithms/imitation/bc/#behavioral-cloning",
"text": "Actions space: Discrete|Continuous",
"title": "Behavioral Cloning"
},
{
"location": "/algorithms/imitation/bc/#network-structure",
"text": "",
"title": "Network Structure"
},
{
"location": "/algorithms/imitation/bc/#algorithm-description",
"text": "",
"title": "Algorithm Description"
},
{
"location": "/algorithms/imitation/bc/#training-the-network",
"text": "The replay buffer contains the expert demonstrations for the task.\nThese demonstrations are given as state, action tuples, and with no reward.\nThe training goal is to reduce the difference between the actions predicted by the network and the actions taken by the expert for each state. Sample a batch of transitions from the replay buffer. Use the current states as input to the network, and the expert actions as the targets of the network. The loss function for the network is MSE, and therefore we use the Q head to minimize this loss.",
"title": "Training the network"
},
{
"location": "/dashboard/",
"text": "Reinforcement learning algorithms are neat. That is - when they work. But when they don't, RL algorithms are often quite tricky to debug. \n\n\nFinding the root cause for why things break in RL is rather difficult. Moreover, different RL algorithms shine in some aspects, but then lack on other. Comparing the algorithms faithfully is also a hard task, which requires the right tools.\n\n\nCoach Dashboard is a visualization tool which simplifies the analysis of the training process. Each run of Coach extracts a lot of information from within the algorithm and stores it in the experiment directory. This information is very valuable for debugging, analyzing and comparing different algorithms. But without a good visualization tool, this information can not be utilized. This is where Coach Dashboard takes place.\n\n\nVisualizing Signals\n\n\nCoach Dashboard exposes a convenient user interface for visualizing the training signals. The signals are dynamically updated - during the agent training. Additionaly, it allows selecting a subset of the available signals, and then overlaying them on top of each other. \n\n\n\n\n\n\n\n\n\n\n\n\n\nHolding the CTRL key, while selecting signals, will allow visualizing more than one signal. \n\n\nSignals can be visualized, using either of the Y-axes, in order to visualize signals with different scales. To move a signal to the second Y-axis, select it and press the 'Toggle Second Axis' button.\n\n\n\n\nTracking Statistics\n\n\nWhen running parallel algorithms, such as A3C, it often helps visualizing the learning of all the workers, at the same time. Coach Dashboard allows viewing multiple signals (and even smooth them out, if required) from multiple workers. In addition, it supports viewing the mean and standard deviation of the same signal, across different workers, using Bollinger bands. \n\n\n\n\n\n\n\n\n\n \n\n \nDisplaying Bollinger Bands\n\n\n\n\n\n \n\n \nDisplaying All The Workers\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\nComparing Runs\n\n\nReinforcement learning algorithms are notoriously known as unstable, and suffer from high run-to-run variance. This makes benchmarking and comparing different algorithms even harder. To ease this process, it is common to execute several runs of the same algorithm and average over them. This is easy to do with Coach Dashboard, by centralizing all the experiment directories in a single directory, and then loading them as a single group. Loading several groups of different algorithms then allows comparing the averaged signals, such as the total episode reward. \n\n\nIn RL, there are several interesting performance metrics to consider, and this is easy to do by controlling the X-axis units in Coach Dashboard. It is possible to switch between several options such as the total number of steps or the total training time.\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\nComparing Several Algorithms According to the Time Passed\n\n\n\n\n\n\n\n\n\n\n\n\n\nComparing Several Algorithms According to the Number of Episodes Played",
"title": "Coach Dashboard"
},
{
"location": "/dashboard/#visualizing-signals",
"text": "Coach Dashboard exposes a convenient user interface for visualizing the training signals. The signals are dynamically updated - during the agent training. Additionaly, it allows selecting a subset of the available signals, and then overlaying them on top of each other. Holding the CTRL key, while selecting signals, will allow visualizing more than one signal. Signals can be visualized, using either of the Y-axes, in order to visualize signals with different scales. To move a signal to the second Y-axis, select it and press the 'Toggle Second Axis' button.",
"title": "Visualizing Signals"
},
{
"location": "/dashboard/#tracking-statistics",
"text": "When running parallel algorithms, such as A3C, it often helps visualizing the learning of all the workers, at the same time. Coach Dashboard allows viewing multiple signals (and even smooth them out, if required) from multiple workers. In addition, it supports viewing the mean and standard deviation of the same signal, across different workers, using Bollinger bands. \n \n Displaying Bollinger Bands \n \n Displaying All The Workers",
"title": "Tracking Statistics"
},
{
"location": "/dashboard/#comparing-runs",
"text": "Reinforcement learning algorithms are notoriously known as unstable, and suffer from high run-to-run variance. This makes benchmarking and comparing different algorithms even harder. To ease this process, it is common to execute several runs of the same algorithm and average over them. This is easy to do with Coach Dashboard, by centralizing all the experiment directories in a single directory, and then loading them as a single group. Loading several groups of different algorithms then allows comparing the averaged signals, such as the total episode reward. In RL, there are several interesting performance metrics to consider, and this is easy to do by controlling the X-axis units in Coach Dashboard. It is possible to switch between several options such as the total number of steps or the total training time. Comparing Several Algorithms According to the Time Passed Comparing Several Algorithms According to the Number of Episodes Played",
"title": "Comparing Runs"
},
{
"location": "/contributing/add_agent/",
"text": "Coach's modularity makes adding an agent a simple and clean task, that involves the following steps:\n\n\n\n\n\n\nImplement your algorithm in a new file. The agent can inherit base classes such as \nValueOptimizationAgent\n or\n \nActorCriticAgent\n, or the more generic \nAgent\n base class.\n\n\n\n\nValueOptimizationAgent\n, \nPolicyOptimizationAgent\n and \nAgent\n are abstract classes. \nlearn_from_batch() should be overriden with the desired behavior for the algorithm being implemented.\nIf deciding to inherit from \nAgent\n, also choose_action() should be overriden.\ndef learn_from_batch(self, batch) -> Tuple[float, List, List]:\n \"\"\"\n Given a batch of transitions, calculates their target values and updates the network.\n :param batch: A list of transitions\n :return: The total loss of the training, the loss per head and the unclipped gradients\n \"\"\"\n\ndef choose_action(self, curr_state):\n \"\"\"\n choose an action to act with in the current episode being played. Different behavior might be exhibited when training\n or testing.\n\n :param curr_state: the current state to act upon.\n :return: chosen action, some action value describing the action (q-value, probability, etc)\n \"\"\"\n\n\n\n\n\n\n\n\n\n\n\nImplement your agent's specific network head, if needed, at the implementation for the framework of your choice.\n For example \narchitectures/neon_components/heads.py\n. The head will inherit the generic base class Head.\n A new output type should be added to configurations.py, and a mapping between the new head and output type should\n be defined in the get_output_head() function at \narchitectures/neon_components/general_network.py\n\n\n\n\n\n\nDefine a new parameters class that inherits AgentParameters.\n The parameters class defines all the hyperparameters for the agent, and is initialized with 4 main components:\n\n\n\n\nalgorithm\n: A class inheriting AlgorithmParameters which defines any algorithm specific parameters\n\n\nexploration\n: A class inheriting ExplorationParameters which defines the exploration policy parameters.\n There are several common exploration policies built-in which you can use, and are defined under\n the exploration sub directory. You can also define your own custom exploration policy.\n\n\nmemory\n: A class inheriting MemoryParameters which defined the memory parameters.\n There are several common memory types built-in which you can use, and are defined under the memories\n sub directory. You can also define your own custom memory.\n\n\nnetworks\n: A dictionary defining all the networks that will be used by the agent. The keys of the dictionary\n define the network name and will be used to access each network through the agent class.\n The dictionary values are a class inheriting NetworkParameters, which define the network structure\n and parameters.\n\n\n\n\nAdditionally, set the path property to return the path to your agent class in the following format:\n\n\n <path to python module>:<name of agent class>\n\n\n\nFor example,\n\n\n class RainbowAgentParameters(AgentParameters):\n def __init__(self):\n super().__init__(algorithm=RainbowAlgorithmParameters(),\n exploration=RainbowExplorationParameters(),\n memory=RainbowMemoryParameters(),\n networks={\"main\": RainbowNetworkParameters()})\n\n @property\n def path(self):\n return 'rainbow.rainbow_agent:RainbowAgent'\n\n\n\n\n\n\n\n(Optional) Define a preset using the new agent type with a given environment, and the hyper-parameters that should\n be used for training on that environment.",
"title": "Adding a New Agent"
},
{
"location": "/contributing/add_env/",
"text": "Adding a new environment to Coach is as easy as solving CartPole. \n\n\nThere are essentially two ways to integrate new environments to Coach:\n\n\nUsing the OpenAI Gym API\n\n\nIf your environment is already using the OpenAI Gym API, you are already good to go.\nWhen selecting the environment parameters in the preset, use GymEnvironmentParameters(),\nand pass the path to your environment source code using the level parameter.\nYou can specify additional parameters for your environment using the additional_simulator_parameters parameter.\nTake for example the definition used in the Pendulum_HAC preset:\n\n\n env_params = GymEnvironmentParameters()\n env_params.level = \"rl_coach.environments.mujoco.pendulum_with_goals:PendulumWithGoals\"\n env_params.additional_simulator_parameters = {\"time_limit\": 1000}\n\n\n\nUsing the Coach API\n\n\nThere are a few simple steps to follow, and we will walk through them one by one.\n\n\n\n\n\n\nCreate a new class for your environment, and inherit the Environment class.\n\n\n\n\n\n\nCoach defines a simple API for implementing a new environment, which are defined in environment/environment.py.\n There are several functions to implement, but only some of them are mandatory.\n\n\nHere are the important ones:\n\n\n def _take_action(self, action_idx: ActionType) -> None:\n \"\"\"\n An environment dependent function that sends an action to the simulator.\n :param action_idx: the action to perform on the environment\n :return: None\n \"\"\"\n\n def _update_state(self) -> None:\n \"\"\"\n Updates the state from the environment.\n Should update self.observation, self.reward, self.done, self.measurements and self.info\n :return: None\n \"\"\"\n\n def _restart_environment_episode(self, force_environment_reset=False) -> None:\n \"\"\"\n Restarts the simulator episode\n :param force_environment_reset: Force the environment to reset even if the episode is not done yet.\n :return: None\n \"\"\"\n\n def _render(self) -> None:\n \"\"\"\n Renders the environment using the native simulator renderer\n :return: None\n \"\"\"\n\n def get_rendered_image(self) -> np.ndarray:\n \"\"\"\n Return a numpy array containing the image that will be rendered to the screen.\n This can be different from the observation. For example, mujoco's observation is a measurements vector.\n :return: numpy array containing the image that will be rendered to the screen\n \"\"\"\n\n\n\n\n\n\n\nCreate a new parameters class for your environment, which inherits the EnvironmentParameters class.\n In the \ninit\n of your class, define all the parameters you used in your Environment class.\n Additionally, fill the path property of the class with the path to your Environment class.\n For example, take a look at the EnvironmentParameters class used for Doom:\n\n\n class DoomEnvironmentParameters(EnvironmentParameters):\n def __init__(self):\n super().__init__()\n self.default_input_filter = DoomInputFilter\n self.default_output_filter = DoomOutputFilter\n self.cameras = [DoomEnvironment.CameraTypes.OBSERVATION]\n\n @property\n def path(self):\n return 'rl_coach.environments.doom_environment:DoomEnvironment'\n\n\n\n\n\n\n\nAnd that's it, you're done. Now just add a new preset with your newly created environment, and start training an agent on top of it.",
"title": "Adding a New Environment"
},
{
"location": "/contributing/add_env/#using-the-openai-gym-api",
"text": "If your environment is already using the OpenAI Gym API, you are already good to go.\nWhen selecting the environment parameters in the preset, use GymEnvironmentParameters(),\nand pass the path to your environment source code using the level parameter.\nYou can specify additional parameters for your environment using the additional_simulator_parameters parameter.\nTake for example the definition used in the Pendulum_HAC preset: env_params = GymEnvironmentParameters()\n env_params.level = \"rl_coach.environments.mujoco.pendulum_with_goals:PendulumWithGoals\"\n env_params.additional_simulator_parameters = {\"time_limit\": 1000}",
"title": "Using the OpenAI Gym API"
},
{
"location": "/contributing/add_env/#using-the-coach-api",
"text": "There are a few simple steps to follow, and we will walk through them one by one. Create a new class for your environment, and inherit the Environment class. Coach defines a simple API for implementing a new environment, which are defined in environment/environment.py.\n There are several functions to implement, but only some of them are mandatory. Here are the important ones: def _take_action(self, action_idx: ActionType) -> None:\n \"\"\"\n An environment dependent function that sends an action to the simulator.\n :param action_idx: the action to perform on the environment\n :return: None\n \"\"\"\n\n def _update_state(self) -> None:\n \"\"\"\n Updates the state from the environment.\n Should update self.observation, self.reward, self.done, self.measurements and self.info\n :return: None\n \"\"\"\n\n def _restart_environment_episode(self, force_environment_reset=False) -> None:\n \"\"\"\n Restarts the simulator episode\n :param force_environment_reset: Force the environment to reset even if the episode is not done yet.\n :return: None\n \"\"\"\n\n def _render(self) -> None:\n \"\"\"\n Renders the environment using the native simulator renderer\n :return: None\n \"\"\"\n\n def get_rendered_image(self) -> np.ndarray:\n \"\"\"\n Return a numpy array containing the image that will be rendered to the screen.\n This can be different from the observation. For example, mujoco's observation is a measurements vector.\n :return: numpy array containing the image that will be rendered to the screen\n \"\"\" Create a new parameters class for your environment, which inherits the EnvironmentParameters class.\n In the init of your class, define all the parameters you used in your Environment class.\n Additionally, fill the path property of the class with the path to your Environment class.\n For example, take a look at the EnvironmentParameters class used for Doom: class DoomEnvironmentParameters(EnvironmentParameters):\n def __init__(self):\n super().__init__()\n self.default_input_filter = DoomInputFilter\n self.default_output_filter = DoomOutputFilter\n self.cameras = [DoomEnvironment.CameraTypes.OBSERVATION]\n\n @property\n def path(self):\n return 'rl_coach.environments.doom_environment:DoomEnvironment' And that's it, you're done. Now just add a new preset with your newly created environment, and start training an agent on top of it.",
"title": "Using the Coach API"
}
]
}
File renamed without changes.
+61 -86
View File
@@ -1,158 +1,133 @@
<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
<url>
<loc>None/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/design/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/usage/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/usage/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/design/features/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/dqn/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/design/control_flow/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/double_dqn/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/design/network/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/dueling_dqn/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/design/filters/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/categorical_dqn/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/dqn/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/mmc/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/double_dqn/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/pal/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/dueling_dqn/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/nec/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/categorical_dqn/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/bs_dqn/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/mmc/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/n_step/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/pal/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/value_optimization/naf/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/nec/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/policy_optimization/pg/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/bs_dqn/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/policy_optimization/ac/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/n_step/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/policy_optimization/ddpg/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/value_optimization/naf/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/policy_optimization/ppo/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/policy_optimization/pg/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/policy_optimization/cppo/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/policy_optimization/ac/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/other/dfp/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/policy_optimization/ddpg/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/algorithms/imitation/bc/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/policy_optimization/ppo/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/dashboard/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/policy_optimization/cppo/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/contributing/add_agent/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/other/dfp/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>None/contributing/add_env/</loc>
<lastmod>2017-12-18</lastmod>
<loc>/algorithms/imitation/bc/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>/dashboard/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>/contributing/add_agent/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
<url>
<loc>/contributing/add_env/</loc>
<lastmod>2018-08-09</lastmod>
<changefreq>daily</changefreq>
</url>
</urlset>
+163 -207
View File
@@ -3,33 +3,29 @@
<!--[if gt IE 8]><!--> <html class="no-js" lang="en" > <!--<![endif]-->
<head>
<meta charset="utf-8">
<meta http-equiv="X-UA-Compatible" content="IE=edge">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Usage - Reinforcement Learning Coach Documentation</title>
<link rel="shortcut icon" href="../img/favicon.ico">
<title>Usage - Reinforcement Learning Coach</title>
<link href='https://fonts.googleapis.com/css?family=Lato:400,700|Roboto+Slab:400,700|Inconsolata:400,700' rel='stylesheet' type='text/css'>
<link rel="stylesheet" href="../css/theme.css" type="text/css" />
<link rel="stylesheet" href="../css/theme_extra.css" type="text/css" />
<link rel="stylesheet" href="../css/highlight.css">
<link href="../extra.css" rel="stylesheet">
<script>
// Current page data
var mkdocs_page_name = "Usage";
var mkdocs_page_input_path = "usage.md";
var mkdocs_page_url = "/usage/";
</script>
<script src="../js/jquery-2.1.1.min.js"></script>
<script src="../js/modernizr-2.8.3.min.js"></script>
<script type="text/javascript" src="../js/highlight.pack.js"></script>
<script src="../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script type="text/javascript" src="../js/highlight.pack.js"></script>
</head>
@@ -40,7 +36,7 @@
<nav data-toggle="wy-nav-shift" class="wy-nav-side stickynav">
<div class="wy-side-nav-search">
<a href="../index.html" class="icon icon-home"> Reinforcement Learning Coach Documentation</a>
<a href=".." class="icon icon-home"> Reinforcement Learning Coach</a>
<div role="search">
<form id ="rtd-search-form" class="wy-form" action="../search.html" method="get">
<input type="text" name="q" placeholder="Search docs" />
@@ -49,205 +45,160 @@
</div>
<div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="main navigation">
<ul class="current">
<ul class="current">
<li>
<li class="toctree-l1 ">
<a class="" href="../index.html">Home</a>
</li>
<li>
<li class="toctree-l1">
<a class="" href="..">Home</a>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../design/index.html">Design</a>
</li>
<li>
<li>
<li class="toctree-l1 current">
<a class="current" href="./index.html">Usage</a>
<ul>
<li class="toctree-l3"><a href="#coach-usage">Coach Usage</a></li>
<li><a class="toctree-l4" href="#training-an-agent">Training an Agent</a></li>
<li><a class="toctree-l4" href="#evaluating-an-agent">Evaluating an Agent</a></li>
<li><a class="toctree-l4" href="#playing-with-the-environment-as-a-human">Playing with the Environment as a Human</a></li>
<li><a class="toctree-l4" href="#learning-through-imitation-learning">Learning Through Imitation Learning</a></li>
<li><a class="toctree-l4" href="#visualizations">Visualizations</a></li>
<li><a class="toctree-l4" href="#switching-between-deep-learning-frameworks">Switching between deep learning frameworks</a></li>
<li><a class="toctree-l4" href="#additional-flags">Additional Flags</a></li>
</ul>
</li>
<li>
<li>
<li class="toctree-l1 current">
<a class="current" href="./">Usage</a>
<ul class="subnav">
<li><span>Algorithms</span></li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/dqn/index.html">DQN</a>
<li class="toctree-l2"><a href="#coach-usage">Coach Usage</a></li>
<ul>
</li>
<li><a class="toctree-l3" href="#training-an-agent">Training an Agent</a></li>
<li><a class="toctree-l3" href="#evaluating-an-agent">Evaluating an Agent</a></li>
<li><a class="toctree-l3" href="#playing-with-the-environment-as-a-human">Playing with the Environment as a Human</a></li>
<li><a class="toctree-l3" href="#learning-through-imitation-learning">Learning Through Imitation Learning</a></li>
<li><a class="toctree-l3" href="#visualizations">Visualizations</a></li>
<li><a class="toctree-l3" href="#switching-between-deep-learning-frameworks">Switching between deep learning frameworks</a></li>
<li><a class="toctree-l3" href="#additional-flags">Additional Flags</a></li>
</ul>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/double_dqn/index.html">Double DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/dueling_dqn/index.html">Dueling DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/categorical_dqn/index.html">Categorical DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/mmc/index.html">Mixed Monte Carlo</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/pal/index.html">Persistent Advantage Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/nec/index.html">Neural Episodic Control</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/bs_dqn/index.html">Bootstrapped DQN</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/n_step/index.html">N-Step Q Learning</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/value_optimization/naf/index.html">Normalized Advantage Functions</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/pg/index.html">Policy Gradient</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ac/index.html">Actor-Critic</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ddpg/index.html">Deep Determinstic Policy Gradients</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/ppo/index.html">Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/policy_optimization/cppo/index.html">Clipped Proximal Policy Optimization</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/other/dfp/index.html">Direct Future Prediction</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../algorithms/imitation/bc/index.html">Behavioral Cloning</a>
</li>
</ul>
<li>
</li>
<li>
<li class="toctree-l1 ">
<a class="" href="../dashboard/index.html">Coach Dashboard</a>
</li>
<li>
<li>
<li class="toctree-l1">
<span class="caption-text">Design</span>
<ul class="subnav">
<li><span>Contributing</span></li>
<li class="toctree-l1 ">
<a class="" href="../contributing/add_agent/index.html">Adding a New Agent</a>
</li>
<li class="toctree-l1 ">
<a class="" href="../contributing/add_env/index.html">Adding a New Environment</a>
</li>
<li class="">
<a class="" href="../design/features/">Features</a>
</li>
<li class="">
<a class="" href="../design/control_flow/">Control Flow</a>
</li>
<li class="">
<a class="" href="../design/network/">Network</a>
</li>
<li class="">
<a class="" href="../design/filters/">Filters</a>
</li>
</ul>
<li>
</li>
<li class="toctree-l1">
<span class="caption-text">Algorithms</span>
<ul class="subnav">
<li class="">
<a class="" href="../algorithms/value_optimization/dqn/">DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/double_dqn/">Double DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/dueling_dqn/">Dueling DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/categorical_dqn/">Categorical DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/mmc/">Mixed Monte Carlo</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/pal/">Persistent Advantage Learning</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/nec/">Neural Episodic Control</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/bs_dqn/">Bootstrapped DQN</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/n_step/">N-Step Q Learning</a>
</li>
<li class="">
<a class="" href="../algorithms/value_optimization/naf/">Normalized Advantage Functions</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/pg/">Policy Gradient</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/ac/">Actor-Critic</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/ddpg/">Deep Determinstic Policy Gradients</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/ppo/">Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../algorithms/policy_optimization/cppo/">Clipped Proximal Policy Optimization</a>
</li>
<li class="">
<a class="" href="../algorithms/other/dfp/">Direct Future Prediction</a>
</li>
<li class="">
<a class="" href="../algorithms/imitation/bc/">Behavioral Cloning</a>
</li>
</ul>
</li>
<li class="toctree-l1">
<a class="" href="../dashboard/">Coach Dashboard</a>
</li>
<li class="toctree-l1">
<span class="caption-text">Contributing</span>
<ul class="subnav">
<li class="">
<a class="" href="../contributing/add_agent/">Adding a New Agent</a>
</li>
<li class="">
<a class="" href="../contributing/add_env/">Adding a New Environment</a>
</li>
</ul>
</li>
</ul>
</div>
@@ -259,7 +210,7 @@
<nav class="wy-nav-top" role="navigation" aria-label="top navigation">
<i data-toggle="wy-nav-top" class="fa fa-bars"></i>
<a href="../index.html">Reinforcement Learning Coach Documentation</a>
<a href="..">Reinforcement Learning Coach</a>
</nav>
@@ -267,7 +218,7 @@
<div class="rst-content">
<div role="navigation" aria-label="breadcrumbs navigation">
<ul class="wy-breadcrumbs">
<li><a href="../index.html">Docs</a> &raquo;</li>
<li><a href="..">Docs</a> &raquo;</li>
@@ -463,10 +414,10 @@ The most up to date description can be found by using the <code>-h</code> flag.<
<div class="rst-footer-buttons" role="navigation" aria-label="footer navigation">
<a href="../algorithms/value_optimization/dqn/index.html" class="btn btn-neutral float-right" title="DQN"/>Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../design/features/" class="btn btn-neutral float-right" title="Features">Next <span class="icon icon-circle-arrow-right"></span></a>
<a href="../design/index.html" class="btn btn-neutral" title="Design"><span class="icon icon-circle-arrow-left"></span> Previous</a>
<a href=".." class="btn btn-neutral" title="Home"><span class="icon icon-circle-arrow-left"></span> Previous</a>
</div>
@@ -480,7 +431,7 @@ The most up to date description can be found by using the <code>-h</code> flag.<
Built with <a href="http://www.mkdocs.org">MkDocs</a> using a <a href="https://github.com/snide/sphinx_rtd_theme">theme</a> provided by <a href="https://readthedocs.org">Read the Docs</a>.
</footer>
</div>
</div>
@@ -488,17 +439,22 @@ The most up to date description can be found by using the <code>-h</code> flag.<
</div>
<div class="rst-versions" role="note" style="cursor: pointer">
<div class="rst-versions" role="note" style="cursor: pointer">
<span class="rst-current-version" data-toggle="rst-current-version">
<span><a href="../design/index.html" style="color: #fcfcfc;">&laquo; Previous</a></span>
<span><a href=".." style="color: #fcfcfc;">&laquo; Previous</a></span>
<span style="margin-left: 15px"><a href="../algorithms/value_optimization/dqn/index.html" style="color: #fcfcfc">Next &raquo;</a></span>
<span style="margin-left: 15px"><a href="../design/features/" style="color: #fcfcfc">Next &raquo;</a></span>
</span>
</div>
<script>var base_url = '..';</script>
<script src="../js/theme.js"></script>
<script src="https://cdn.mathjax.org/mathjax/latest/MathJax.js?config=TeX-AMS_HTML"></script>
<script src="../search/require.js"></script>
<script src="../search/search.js"></script>
</body>
</html>