Batch RL (#238)

2026-04-26 19:01:28 +02:00 · 2019-03-19 18:07:09 +02:00
parent 4a8451ff02
commit e3c7e526c7
38 changed files with 1003 additions and 87 deletions
@@ -64,6 +64,9 @@ class BootstrappedDQNAgent(ValueOptimizationAgent):
        q_st_plus_1 = result[:self.ap.exploration.architecture_num_q_heads]
        TD_targets = result[self.ap.exploration.architecture_num_q_heads:]

+        # add Q value samples for logging
+        self.q_values.add_sample(TD_targets)
+
        # initialize with the current prediction so that we will
        #  only update the action that we have actually done in this transition
        for i in range(self.ap.network_wrappers['main'].batch_size):