last questions

This commit is contained in:
Arjun Patel
2019-03-07 19:07:42 -08:00
parent 3cbe8b49a2
commit ea6cf86eab
8 changed files with 86 additions and 757 deletions
+1 -1
View File
@@ -68,7 +68,7 @@ def question3e():
def question8():
answerEpsilon = None
answerLearningRate = None
return answerEpsilon, answerLearningRate
return 'NOT POSSIBLE'
# If not possible, return 'NOT POSSIBLE'
+79 -14
View File
@@ -16,7 +16,10 @@ from game import *
from learningAgents import ReinforcementAgent
from featureExtractors import *
import random,util,math
import random
import util
import math
class QLearningAgent(ReinforcementAgent):
"""
@@ -38,11 +41,18 @@ class QLearningAgent(ReinforcementAgent):
- self.getLegalActions(state)
which returns legal actions for a state
"""
def __init__(self, **args):
"You can initialize Q-values here..."
ReinforcementAgent.__init__(self, **args)
"*** YOUR CODE HERE ***"
self.values = util.Counter()
# self.mdp = mdp
# self.discount = discount
# self.iterations = iterations
# self.values = util.Counter() # A Counter is a dict with default 0
# self.runValueIteration()
def getQValue(self, state, action):
"""
@@ -51,8 +61,8 @@ class QLearningAgent(ReinforcementAgent):
or the Q node value otherwise
"""
"*** YOUR CODE HERE ***"
util.raiseNotDefined()
t = (state, action)
return self.values[t]
def computeValueFromQValues(self, state):
"""
@@ -62,7 +72,18 @@ class QLearningAgent(ReinforcementAgent):
terminal state, you should return a value of 0.0.
"""
"*** YOUR CODE HERE ***"
util.raiseNotDefined()
INF, NEG_INF = float("inf"), -float("inf")
options_actions = self.getLegalActions(state)
optimal = NEG_INF
for action in options_actions:
if self.getQValue(state, action) > optimal:
optimal = self.getQValue(state, action)
if optimal != NEG_INF:
return optimal
else:
return 0.0
def computeActionFromQValues(self, state):
"""
@@ -71,7 +92,16 @@ class QLearningAgent(ReinforcementAgent):
you should return None.
"""
"*** YOUR CODE HERE ***"
util.raiseNotDefined()
# Exit/no actions to perform
if len(self.getLegalActions(state)) == 0:
return None
optimal = self.computeValueFromQValues(state)
policy = [action for action in self.getLegalActions(
state) if optimal == self.getQValue(state, action)]
# grab an action
return random.choice(policy)
def getAction(self, state):
"""
@@ -85,10 +115,14 @@ class QLearningAgent(ReinforcementAgent):
HINT: To pick randomly from a list, use random.choice(list)
"""
# Pick Action
legalActions = self.getLegalActions(state)
action = None
"*** YOUR CODE HERE ***"
util.raiseNotDefined()
options_actions = self.getLegalActions(state)
action = None
if util.flipCoin(self.epsilon):
action = random.choice(options_actions)
else:
action = self.computeActionFromQValues(state)
return action
@@ -102,7 +136,13 @@ class QLearningAgent(ReinforcementAgent):
it will be called on your behalf
"""
"*** YOUR CODE HERE ***"
util.raiseNotDefined()
t = (state, action)
prev = self.values[t]
val = reward + \
(self.discount * self.computeValueFromQValues(nextState))
self.values[t] = (1 - self.alpha) * \
prev + self.alpha * val
def getPolicy(self, state):
return self.computeActionFromQValues(state)
@@ -114,7 +154,7 @@ class QLearningAgent(ReinforcementAgent):
class PacmanQAgent(QLearningAgent):
"Exactly the same as QLearningAgent, but with different default parameters"
def __init__(self, epsilon=0.05,gamma=0.8,alpha=0.2, numTraining=0, **args):
def __init__(self, epsilon=0.05, gamma=0.8, alpha=0.2, numTraining=0, **args):
"""
These default parameters can be changed from the pacman.py command line.
For example, to change the exploration rate, try:
@@ -138,8 +178,8 @@ class PacmanQAgent(QLearningAgent):
informs parent of action for Pacman. Do not change or remove this
method.
"""
action = QLearningAgent.getAction(self,state)
self.doAction(state,action)
action = QLearningAgent.getAction(self, state)
self.doAction(state, action)
return action
@@ -151,6 +191,7 @@ class ApproximateQAgent(PacmanQAgent):
and update. All other QLearningAgent functions
should work as is.
"""
def __init__(self, extractor='IdentityExtractor', **args):
self.featExtractor = util.lookup(extractor, globals())()
PacmanQAgent.__init__(self, **args)
@@ -165,14 +206,35 @@ class ApproximateQAgent(PacmanQAgent):
where * is the dotProduct operator
"""
"*** YOUR CODE HERE ***"
util.raiseNotDefined()
# model_features = self.featExtractor.getFeatures(state, action)
# return sum([self.weights[feat] * val for feat, val in model_features.iteritems()])
model_features = self.featExtractor.getFeatures(state, action)
return sum([model_features[feat] * self.weights[feat] for feat in model_features])
def update(self, state, action, nextState, reward):
"""
Should update your weights based on transition
"""
"*** YOUR CODE HERE ***"
util.raiseNotDefined()
# val = reward + self.discount * self.computeValueFromQValues(nextState)
# prev = self.getQValue(state, action)
# diff = val - prev
# model_features = self.featExtractor.getFeatures(state, action)
# for feat, val in model_features.iteritems():
# self.weights[feat] += self.alpha * \
# diff * model_features[feat]
discounted = self.discount * self.getValue(nextState)
diff = (reward + discounted) - \
self.getQValue(state, action)
model_features = self.featExtractor.getFeatures(state, action)
for feat in model_features:
self.weights[feat] = self.weights[feat] + \
self.alpha * diff * model_features[feat]
def final(self, state):
"Called at the end of each game."
@@ -183,4 +245,7 @@ class ApproximateQAgent(PacmanQAgent):
if self.episodesSoFar == self.numTraining:
# you might want to print your weights here for debugging
"*** YOUR CODE HERE ***"
# print(self.weights)
# for i in features:
# print(self.weights[i], i)
pass
File diff suppressed because one or more lines are too long
-150
View File
@@ -1,150 +0,0 @@
Values at iteration 0 are correct.
Student/correct solution:
values_k_0: """
0.0000 0.0000 0.0000
"""
Q-Values at iteration 0 for action south are correct.
Student/correct solution:
q_values_k_0_action_south: """
illegal 0.0000 illegal
"""
Q-Values at iteration 0 for action north are correct.
Student/correct solution:
q_values_k_0_action_north: """
illegal 0.0000 illegal
"""
Q-Values at iteration 0 for action exit are correct.
Student/correct solution:
q_values_k_0_action_exit: """
10.0000 illegal -10.0000
"""
Q-Values at iteration 0 for action west are correct.
Student/correct solution:
q_values_k_0_action_west: """
illegal 0.0000 illegal
"""
Q-Values at iteration 0 for action east are correct.
Student/correct solution:
q_values_k_0_action_east: """
illegal 0.0000 illegal
"""
Values at iteration 1 are correct.
Student/correct solution:
values_k_1: """
10.0000 0.0000 0.0000
"""
Q-Values at iteration 1 for action south are correct.
Student/correct solution:
q_values_k_1_action_south: """
illegal 0.0000 illegal
"""
Q-Values at iteration 1 for action north are correct.
Student/correct solution:
q_values_k_1_action_north: """
illegal 0.0000 illegal
"""
Q-Values at iteration 1 for action exit are correct.
Student/correct solution:
q_values_k_1_action_exit: """
10.0000 illegal -10.0000
"""
Q-Values at iteration 1 for action west are correct.
Student/correct solution:
q_values_k_1_action_west: """
illegal 5.0000 illegal
"""
Q-Values at iteration 1 for action east are correct.
Student/correct solution:
q_values_k_1_action_east: """
illegal 0.0000 illegal
"""
Values at iteration 2 are NOT correct.
Student solution:
values_k_2: """
10.0000 5.0000 0.0000
"""
Correct solution:
values_k_2: """
10.0000 0.0000 -10.0000
"""
Q-Values at iteration 2 for action south are NOT correct.
Student solution:
q_values_k_2_action_south: """
illegal 2.5000 illegal
"""
Correct solution:
q_values_k_2_action_south: """
illegal 0.0000 illegal
"""
Q-Values at iteration 2 for action north are NOT correct.
Student solution:
q_values_k_2_action_north: """
illegal 2.5000 illegal
"""
Correct solution:
q_values_k_2_action_north: """
illegal 0.0000 illegal
"""
Q-Values at iteration 2 for action exit are correct.
Student/correct solution:
q_values_k_2_action_exit: """
10.0000 illegal -10.0000
"""
Q-Values at iteration 2 for action west are correct.
Student/correct solution:
q_values_k_2_action_west: """
illegal 5.0000 illegal
"""
Q-Values at iteration 2 for action east are NOT correct.
Student solution:
q_values_k_2_action_east: """
illegal 0.0000 illegal
"""
Correct solution:
q_values_k_2_action_east: """
illegal -5.0000 illegal
"""
-156
View File
@@ -1,156 +0,0 @@
Values at iteration 0 are correct.
Student/correct solution:
values_k_0: """
0.0000 0.0000 0.0000
"""
Q-Values at iteration 0 for action south are correct.
Student/correct solution:
q_values_k_0_action_south: """
illegal 0.0000 illegal
"""
Q-Values at iteration 0 for action north are correct.
Student/correct solution:
q_values_k_0_action_north: """
illegal 0.0000 illegal
"""
Q-Values at iteration 0 for action exit are correct.
Student/correct solution:
q_values_k_0_action_exit: """
10.0000 illegal -10.0000
"""
Q-Values at iteration 0 for action west are correct.
Student/correct solution:
q_values_k_0_action_west: """
illegal 0.0000 illegal
"""
Q-Values at iteration 0 for action east are correct.
Student/correct solution:
q_values_k_0_action_east: """
illegal 0.0000 illegal
"""
Values at iteration 1 are correct.
Student/correct solution:
values_k_1: """
10.0000 0.0000 0.0000
"""
Q-Values at iteration 1 for action south are correct.
Student/correct solution:
q_values_k_1_action_south: """
illegal 0.9375 illegal
"""
Q-Values at iteration 1 for action north are correct.
Student/correct solution:
q_values_k_1_action_north: """
illegal 0.9375 illegal
"""
Q-Values at iteration 1 for action exit are correct.
Student/correct solution:
q_values_k_1_action_exit: """
10.0000 illegal -10.0000
"""
Q-Values at iteration 1 for action west are correct.
Student/correct solution:
q_values_k_1_action_west: """
illegal 5.6250 illegal
"""
Q-Values at iteration 1 for action east are correct.
Student/correct solution:
q_values_k_1_action_east: """
illegal 0.0000 illegal
"""
Values at iteration 2 are NOT correct.
Student solution:
values_k_2: """
10.0000 5.6250 0.0000
"""
Correct solution:
values_k_2: """
10.0000 0.0000 -10.0000
"""
Q-Values at iteration 2 for action south are NOT correct.
Student solution:
q_values_k_2_action_south: """
illegal 4.1016 illegal
"""
Correct solution:
q_values_k_2_action_south: """
illegal 0.0000 illegal
"""
Q-Values at iteration 2 for action north are NOT correct.
Student solution:
q_values_k_2_action_north: """
illegal 4.1016 illegal
"""
Correct solution:
q_values_k_2_action_north: """
illegal 0.0000 illegal
"""
Q-Values at iteration 2 for action exit are correct.
Student/correct solution:
q_values_k_2_action_exit: """
10.0000 illegal -10.0000
"""
Q-Values at iteration 2 for action west are NOT correct.
Student solution:
q_values_k_2_action_west: """
illegal 6.6797 illegal
"""
Correct solution:
q_values_k_2_action_west: """
illegal 5.6250 illegal
"""
Q-Values at iteration 2 for action east are NOT correct.
Student solution:
q_values_k_2_action_east: """
illegal 1.0547 illegal
"""
Correct solution:
q_values_k_2_action_east: """
illegal -5.6250 illegal
"""
-194
View File
@@ -1,194 +0,0 @@
Values at iteration 0 are correct.
Student/correct solution:
values_k_0: """
__________ 0.0000 0.0000 0.0000 0.0000 0.0000 __________
0.0000 0.0000 0.0000 0.0000 0.0000 0.0000 0.0000
__________ 0.0000 0.0000 0.0000 0.0000 0.0000 __________
"""
Q-Values at iteration 0 for action south are correct.
Student/correct solution:
q_values_k_0_action_south: """
__________ illegal illegal illegal illegal illegal __________
illegal 0.0000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 0 for action north are correct.
Student/correct solution:
q_values_k_0_action_north: """
__________ illegal illegal illegal illegal illegal __________
illegal 0.0000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 0 for action exit are correct.
Student/correct solution:
q_values_k_0_action_exit: """
__________ -100.0000 -100.0000 -100.0000 -100.0000 -100.0000 __________
1.0000 illegal illegal illegal illegal illegal 10.0000
__________ -100.0000 -100.0000 -100.0000 -100.0000 -100.0000 __________
"""
Q-Values at iteration 0 for action west are correct.
Student/correct solution:
q_values_k_0_action_west: """
__________ illegal illegal illegal illegal illegal __________
illegal 0.0000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 0 for action east are correct.
Student/correct solution:
q_values_k_0_action_east: """
__________ illegal illegal illegal illegal illegal __________
illegal 0.0000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Values at iteration 1 are correct.
Student/correct solution:
values_k_1: """
__________ 0.0000 0.0000 0.0000 0.0000 0.0000 __________
0.0000 0.0000 0.0000 0.0000 0.0000 0.0000 0.0000
__________ -100.0000 0.0000 0.0000 0.0000 0.0000 __________
"""
Q-Values at iteration 1 for action south are correct.
Student/correct solution:
q_values_k_1_action_south: """
__________ illegal illegal illegal illegal illegal __________
illegal -76.5000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 1 for action north are correct.
Student/correct solution:
q_values_k_1_action_north: """
__________ illegal illegal illegal illegal illegal __________
illegal 0.0000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 1 for action exit are correct.
Student/correct solution:
q_values_k_1_action_exit: """
__________ -100.0000 -100.0000 -100.0000 -100.0000 -100.0000 __________
1.0000 illegal illegal illegal illegal illegal 10.0000
__________ -100.0000 -100.0000 -100.0000 -100.0000 -100.0000 __________
"""
Q-Values at iteration 1 for action west are correct.
Student/correct solution:
q_values_k_1_action_west: """
__________ illegal illegal illegal illegal illegal __________
illegal -4.2500 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 1 for action east are correct.
Student/correct solution:
q_values_k_1_action_east: """
__________ illegal illegal illegal illegal illegal __________
illegal -4.2500 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Values at iteration 2 are NOT correct.
Student solution:
values_k_2: """
__________ 0.0000 0.0000 0.0000 0.0000 0.0000 __________
0.0000 0.0000 0.0000 0.0000 0.0000 0.0000 0.0000
__________ -100.0000 0.0000 0.0000 0.0000 0.0000 __________
"""
Correct solution:
values_k_2: """
__________ -100.0000 0.0000 0.0000 0.0000 0.0000 __________
0.0000 0.0000 0.0000 0.0000 0.0000 0.0000 0.0000
__________ -100.0000 0.0000 0.0000 0.0000 0.0000 __________
"""
Q-Values at iteration 2 for action south are correct.
Student/correct solution:
q_values_k_2_action_south: """
__________ illegal illegal illegal illegal illegal __________
illegal -76.5000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 2 for action north are NOT correct.
Student solution:
q_values_k_2_action_north: """
__________ illegal illegal illegal illegal illegal __________
illegal 0.0000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Correct solution:
q_values_k_2_action_north: """
__________ illegal illegal illegal illegal illegal __________
illegal -76.5000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 2 for action exit are correct.
Student/correct solution:
q_values_k_2_action_exit: """
__________ -100.0000 -100.0000 -100.0000 -100.0000 -100.0000 __________
1.0000 illegal illegal illegal illegal illegal 10.0000
__________ -100.0000 -100.0000 -100.0000 -100.0000 -100.0000 __________
"""
Q-Values at iteration 2 for action west are NOT correct.
Student solution:
q_values_k_2_action_west: """
__________ illegal illegal illegal illegal illegal __________
illegal -4.2500 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Correct solution:
q_values_k_2_action_west: """
__________ illegal illegal illegal illegal illegal __________
illegal -8.5000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Q-Values at iteration 2 for action east are NOT correct.
Student solution:
q_values_k_2_action_east: """
__________ illegal illegal illegal illegal illegal __________
illegal -4.2500 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
Correct solution:
q_values_k_2_action_east: """
__________ illegal illegal illegal illegal illegal __________
illegal -8.5000 0.0000 0.0000 0.0000 0.0000 illegal
__________ illegal illegal illegal illegal illegal __________
"""
-238
View File
@@ -1,238 +0,0 @@
Values at iteration 0 are correct.
Student/correct solution:
values_k_0: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 __________ 0.0000
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 0.0000 0.0000 0.0000 0.0000
"""
Q-Values at iteration 0 for action south are correct.
Student/correct solution:
q_values_k_0_action_south: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
0.0000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 0 for action north are correct.
Student/correct solution:
q_values_k_0_action_north: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
0.0000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 0 for action exit are correct.
Student/correct solution:
q_values_k_0_action_exit: """
illegal illegal illegal illegal illegal
illegal __________ illegal illegal illegal
illegal __________ 1.0000 __________ 10.0000
illegal illegal illegal illegal illegal
-10.0000 -10.0000 -10.0000 -10.0000 -10.0000
"""
Q-Values at iteration 0 for action west are correct.
Student/correct solution:
q_values_k_0_action_west: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
0.0000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 0 for action east are correct.
Student/correct solution:
q_values_k_0_action_east: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
0.0000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Values at iteration 1 are correct.
Student/correct solution:
values_k_1: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 __________ 0.0000
0.0000 0.0000 0.0000 0.0000 0.0000
-10.0000 0.0000 0.0000 0.0000 0.0000
"""
Q-Values at iteration 1 for action south are correct.
Student/correct solution:
q_values_k_1_action_south: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-7.2000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 1 for action north are correct.
Student/correct solution:
q_values_k_1_action_north: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
0.0000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 1 for action exit are correct.
Student/correct solution:
q_values_k_1_action_exit: """
illegal illegal illegal illegal illegal
illegal __________ illegal illegal illegal
illegal __________ 1.0000 __________ 10.0000
illegal illegal illegal illegal illegal
-10.0000 -10.0000 -10.0000 -10.0000 -10.0000
"""
Q-Values at iteration 1 for action west are correct.
Student/correct solution:
q_values_k_1_action_west: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-0.9000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 1 for action east are correct.
Student/correct solution:
q_values_k_1_action_east: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-0.9000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Values at iteration 2 are NOT correct.
Student solution:
values_k_2: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 __________ 0.0000
0.0000 0.0000 0.0000 0.0000 0.0000
-10.0000 0.0000 0.0000 0.0000 0.0000
"""
Correct solution:
values_k_2: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 __________ 0.0000
0.0000 0.0000 0.0000 0.0000 0.0000
-10.0000 -10.0000 0.0000 0.0000 0.0000
"""
Q-Values at iteration 2 for action south are NOT correct.
Student solution:
q_values_k_2_action_south: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-7.2000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Correct solution:
q_values_k_2_action_south: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-7.2000 -7.2000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 2 for action north are correct.
Student/correct solution:
q_values_k_2_action_north: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
0.0000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 2 for action exit are correct.
Student/correct solution:
q_values_k_2_action_exit: """
illegal illegal illegal illegal illegal
illegal __________ illegal illegal illegal
illegal __________ 1.0000 __________ 10.0000
illegal illegal illegal illegal illegal
-10.0000 -10.0000 -10.0000 -10.0000 -10.0000
"""
Q-Values at iteration 2 for action west are NOT correct.
Student solution:
q_values_k_2_action_west: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-0.9000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Correct solution:
q_values_k_2_action_west: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-0.9000 -0.9000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Q-Values at iteration 2 for action east are NOT correct.
Student solution:
q_values_k_2_action_east: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-0.9000 0.0000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
Correct solution:
q_values_k_2_action_east: """
0.0000 0.0000 0.0000 0.0000 0.0000
0.0000 __________ 0.0000 0.0000 0.0000
0.0000 __________ illegal __________ illegal
-0.9000 -0.9000 0.0000 0.0000 0.0000
illegal illegal illegal illegal illegal
"""
+4 -3
View File
@@ -223,6 +223,7 @@ class PrioritizedSweepingValueIterationAgent(AsynchronousValueIterationAgent):
# computing the predecssors for all states
# For each non-terminal state, do:
# breaking in 2 stages
for curr_state in mdp_states:
# exit the iteration
@@ -280,10 +281,10 @@ class PrioritizedSweepingValueIterationAgent(AsynchronousValueIterationAgent):
if self.mdp.isTerminal(prev):
continue
options_actions = self.mdp.getPossibleActions(curr_state)
optimal = max([self.getQValue(curr_state, x)
options_actions = self.mdp.getPossibleActions(prev)
optimal = max([self.getQValue(prev, x)
for x in options_actions])
# finding -diff
diff = abs(optimal - self.values[prev])
# difference large enough?
if diff > self.theta:
hinge.update(prev, -diff)