import turtle import random WIN_SCORE = 5 GAME_OVER = False creaturescore = 0 creaturebscore = 0 collideok=False collideokb=False #setting bounds for later track generation LEFT_BOUND = -300 RIGHT_BOUND = 350 TOP_BOUND = 300 BOTTOM_BOUND = -300 #generating turtles for texts s = turtle.Turtle() s.penup() s.hideturtle() onescore = 0 enemyprox = "0" enemyprox2 = "0" us = turtle.Turtle() us.hideturtle() us.penup() us.color("black") u = turtle.Turtle() u.hideturtle() u.penup() u.color("black") s = turtle.Turtle() s.hideturtle() s.penup() s.color("black") #start. Optional input of entering data to build a pre-learned Agent while True: try: ask = int(input("agent 1 or 2: ")) if ask == 1 or 2: break else: print("aight thats it we skippin") except ValueError: print("NAH") if ask == 1: raw = input("Paste Intelligence_1_DBID (or press Enter to start fresh): ") try: if raw.strip(): Q = eval(raw) if not isinstance(Q, dict): raise ValueError("Data is not a dictionary") else: Q = {} except Exception as e: print("⚠️ Invalid Intelligence_1_DBID. Starting fresh.") print("Reason:", e) Q = {} Qb = {} elif ask == 2: raw = input("Paste Intelligence_2_DBID (or press Enter to start fresh): ") try: if raw.strip(): Qb = eval(raw) if not isinstance(Qb, dict): raise ValueError("Data is not a dictionary") else: Qb = {} except Exception as e: print("⚠️ Invalid Intelligence_2_DBID. Starting fresh.") print("Reason:", e) Qb = {} Q = {} #making list of datas trackcor = [] enemies = [] #setting area boundaries def wall_proximity(x, y): d_outer = TRACK_OUTER - max(abs(x), abs(y)) d_inner = max(abs(x), abs(y)) - TRACK_INNER return min(d_outer, d_inner) #importing system into python for later Q-learning import numpy as np #background #start... As you can read import turtle def show_start_screen(start_callback): screen = turtle.Screen() screen.setup(600, 600) screen.bgcolor("gray") screen.title("AI Learning Game") screen.tracer(0) # Title text title = turtle.Turtle() title.hideturtle() title.penup() title.color("black") title.goto(0, 100) title.write( "SUPER AI LEARNING GAME", align="center", font=("Arial", 26, "bold") ) # Subtitle subtitle = turtle.Turtle() subtitle.hideturtle() subtitle.penup() subtitle.color("black") subtitle.goto(0, 50) subtitle.write("""Just some stupid game here. You're not supposed to race with them but is supposed to just watch whether which learns faster... Well yeah... Click START to begin""",align="center",font=("Arial", 14, "normal")) # Start button b = turtle.Turtle() b.shape("square") b.color("black", "lightgreen") b.shapesize(stretch_wid=2.5, stretch_len=6) b.penup() b.goto(0, 0) bu = turtle.Turtle() bu.hideturtle() bu.penup() bu.color("black") bu.goto(0, -65) b.write("START",align="center",font=("Arial", 18, "bold")) def clicked(x, y): title.clear() subtitle.clear() bu.clear() b.clear() b.hideturtle() screen.bgcolor("white") screen.update() start_callback() #MAIN GAME STARTS HERE b.onclick(clicked) screen.update() #gemerate enemies def minesotapineapple(): for _ in range(30): p = turtle.Turtle() p.penup() p.shape("triangle") p.goto(random.randint(-250, 250), random.randint(-250, 250)) enemies.append(p) # detect collision between enemies and agents def enemy_collision(): global prev_distance, collideok #for p in enemies... You wanna apply to all ps for p in enemies: if creature.distance(p) <15: if collideok: reset_agents(creature) return True return False #track outer and inner boundaries. Ths is not abandoned unlike the other boundaries. TRACK_OUTER = 260 TRACK_INNER = 180 #This game is a 5 round. You use this at the start too though def reset_round(): global prev_distance, prev_distance_b reset_agents(creature) reset_agentsb(creatureb) placefoodontrack(food) #I will only mention this here but everytime the environment changes, you set the distance between target and agent again. prev_distance = None prev_distance_b = None #Yeah, read the function name def drawtrack(): t = turtle.Turtle() t.hideturtle() t.speed(0) t.penup() # Outer square t.goto(-TRACK_OUTER, -TRACK_OUTER) t.pendown() t.pensize(6) for _ in range(4): t.forward(TRACK_OUTER * 2) t.left(90) # Inner square t.penup() t.goto(-TRACK_INNER, -TRACK_INNER) t.pendown() for _ in range(4): t.forward(TRACK_INNER * 2) t.left(90) t.penup() #check if agents are on track or not def on_track(x, y): """Returns True if position is inside track corridor""" outer_ok = abs(x) < TRACK_OUTER and abs(y) < TRACK_OUTER inner_block = abs(x) < TRACK_INNER and abs(y) < TRACK_INNER return outer_ok and not inner_block def track_collision(turtle_obj): x, y = turtle_obj.xcor(), turtle_obj.ycor() return not on_track(x, y) #there are all respawning and re-generation codes def reset_agents(creature): # Opposite sides of track creature.goto(-220, 220) creature.setheading(270) def reset_agentsb(creatureb): creatureb.goto(-220, -220) creatureb.setheading(90) def placefoodontrack(food): if random.random() < 0.5: food.goto(220, 0) else: food.goto(-220, 0) #these are mostly abandoned. Some are for agent a lastlastaction = None lastaction = None prev_distance = None epsilon = 1 lastlastaction2 = None lastaction2 = None prev_distance2 = None screen = turtle.Screen() screen.setup(600,600) screen.tracer(0) creature = turtle.Turtle() creature.speed(0) creature.penup() creature.shape("circle") food = turtle.Turtle() food.penup() food.shape("square") food.color("blue") screen = turtle.Screen() screen.title("TEST") screen.tracer(0) screen.listen() actions = ["up", "down", "left", "right"] actionsb = ["up2", "down2", "left2", "right2"] keys_pressed = {"w": False, "s": False, "a": False, "d": False} #yes, import numpy for Q-learning. import numpy as np #So you make neurons here based on how much parameters of stste the agent gets via other functions class DQN: def __init__(self): #9 on the code down is the number of neurons coded self.W1 = np.random.randn(9, 32) * 0.1 self.W2 = np.random.randn(32, 4) * 0.1 self.lr = 0.003 #This is the active learning part. Where the agent learns what did what and resulted into what. Its like a memorization part def forward(self, s): h = np.tanh(np.dot(s, self.W1)) return np.dot(h, self.W2), h #And then the agent gets trained based on reward, state, and actions. They compare and connect parameters given and see the independent weights of each connections def train(self, s, target): q, h = self.forward(s) error = target - q #based on the weights/importance of the connection the neurons make, it learns and predicts what the best action is on a certain situation. self.W2 += self.lr * np.outer(h, error) dh = np.dot(error, self.W2.T) * (1 - h**2) self.W1 += self.lr * np.outer(s, dh) dqn = DQN() def get_state_numeric(): #so they get their independent state here. #These parameter/states have their own weights set. Weights are basically how important these factors are dx = (food.xcor() - creature.xcor()) / 600 dy = (food.ycor() - creature.ycor()) / 600 wx = creature.xcor() / 300 wy = creature.ycor() / 300 nearest_enemy = min(enemies, key=lambda p: creature.distance(p)) ex = (nearest_enemy.xcor() - creature.xcor()) / 600 ey = (nearest_enemy.ycor() - creature.ycor()) / 600 wall_dist = wall_proximity(creature.xcor(), creature.ycor()) / 80 heading = creature.heading() / 360 dist_e = creature.distance(nearest_enemy) / 300 return np.array([dx, dy, ex, ey, dist_e, wx,wy, wall_dist, heading]) #yup, now it chooses the best action def choose_action_nn(state): if random.random() < epsilon: return random.randint(0, 3) q_vals, _ = dqn.forward(state) return np.argmax(q_vals) #and they move now here. def do_action(a): nearest_enemy = min(enemies, key=lambda p: creature.distance(p)) dist_e = creature.distance(nearest_enemy) if a == "up": creature.setheading(90) if a == "down": creature.setheading(270) if a == "left": creature.setheading(180) if a == "right": creature.setheading(0) speed = 3 if dist_e < 60 else 6 creature.forward(speed) x,y = creature.xcor(), creature.ycor() #reward functions here. The agents gets positive or negative rewards based on their actions on different situations. #The agents learn whats bad based on how much rewards it recieved. def get_reward(ec, action): global prev_distance, epsilon, enemyprox, onescore, creaturescore reward = 0.0 x,y = creature.xcor(), creature.ycor() dist = creature.distance(food) reward += 0.01 if lastaction == actions[action]: reward -= 0.02 dx = (food.xcor() - creature.xcor()) dy = (food.ycor() - creature.ycor()) wall_dist = wall_proximity(creature.xcor(), creature.ycor()) if wall_dist < 40: reward -= (40 - wall_dist) / 40 * 2 if prev_distance is None: prev_distance = dist return 0 if abs(dx) < 0.01 and abs(dy) < 0.01: reward -= 0.2 nearest_enemy = min(enemies, key=lambda p: creature.distance(p)) dist_e = creature.distance(nearest_enemy) if dist_e < 150: reward -= (150 - dist_e) / 150 * 3 if prev_distance is not None: reward += (prev_distance - dist) * 0.5 x = random.randint(-TRACK_OUTER + 20, TRACK_OUTER - 20) y = random.randint(-TRACK_OUTER + 20, TRACK_OUTER - 20) if dist < 20: creaturescore += 1 reset_round() return +35 prev_distance = dist return reward #Agent updates its brain with state, reward, and a new table for datas def update_nn(state, action, reward, next_state): q_vals, _ = dqn.forward(state) next_q, _ = dqn.forward(next_state) target = q_vals.copy() target[action] = reward + 0.9 * np.max(next_q) target = np.clip(target, -50, 50) dqn.train(state, target) #same thing repeated here def enemy_collision_b(): global prev_distance_b, collideokb for p in enemies: if creatureb.distance(p) < 15: if collideokb: reset_agentsb(creatureb) return True return False lastlastaction_b = None lastaction_b = None prev_distance_b = None epsilon_b = 1 lastlastaction2_b = None lastaction2_b = None prev_distance2_b = None screenb = turtle.Screen() screenb.setup(600, 600) screenb.tracer(0) creatureb = turtle.Turtle() creatureb.speed(0) creatureb.penup() creatureb.shape("circle") creatureb.color("red") actions_b = ["up", "down", "left", "right"] class DQN_b: def __init__(self): self.W1 = np.random.randn(9, 32) * 0.1 self.W2 = np.random.randn(32, 4) * 0.1 self.lr = 0.003 def forward(self, s): h = np.tanh(np.dot(s, self.W1)) return np.dot(h, self.W2), h def train(self, s, target): q, h = self.forward(s) error = target - q self.W2 += self.lr * np.outer(h, error) dh = np.dot(error, self.W2.T) * (1 - h**2) self.W1 += self.lr * np.outer(s, dh) dqnb = DQN_b() def get_state_numeric_b(): dx = (food.xcor() - creatureb.xcor()) / 600 dy = (food.ycor() - creatureb.ycor()) / 600 wx = creatureb.xcor() / 300 wy = creatureb.ycor() / 300 nearest_enemy = min(enemies, key=lambda p: creatureb.distance(p)) ex = (nearest_enemy.xcor() - creatureb.xcor()) / 600 ey = (nearest_enemy.ycor() - creatureb.ycor()) / 600 dist_e = creatureb.distance(nearest_enemy) / 300 heading = creatureb.heading() / 360 wall_dist = wall_proximity(creatureb.xcor(), creatureb.ycor()) / 80 return np.array([dx, dy, ex, ey, dist_e, wx, wy, wall_dist, heading]) def choose_action_nn_b(state): if random.random() < epsilon_b: return random.randint(0, 3) q_vals, _ = dqnb.forward(state) return np.argmax(q_vals) def do_action_b(a): nearest_enemy = min(enemies, key=lambda p: creatureb.distance(p)) dist_e = creatureb.distance(nearest_enemy) if a == "up": creatureb.setheading(90) if a == "down": creatureb.setheading(270) if a == "left": creatureb.setheading(180) if a == "right": creatureb.setheading(0) speed = 3 if dist_e < 60 else 6 creatureb.forward(speed) def get_reward_b(ec, action): global prev_distance_b, epsilon_b, enemyprox2, onescore, creaturebscore reward = 0.0 x, y = creatureb.xcor(), creatureb.ycor() dist = creatureb.distance(food) reward+=0.01 dx = (food.xcor() - creatureb.xcor()) dy = (food.ycor() - creatureb.ycor()) if lastaction_b == actions_b[action]: reward -= 0.02 if abs(dx) < 0.01 and abs(dy) < 0.01: reward -= 0.2 wall_dist = wall_proximity(creatureb.xcor(), creatureb.ycor()) x = random.randint(-TRACK_OUTER + 20, TRACK_OUTER - 20) y = random.randint(-TRACK_OUTER + 20, TRACK_OUTER - 20) if wall_dist < 40: reward -= (40 - wall_dist) / 40 * 2 if prev_distance_b is None: prev_distance_b = dist return 0 nearest_enemy = min(enemies, key=lambda p: creatureb.distance(p)) dist_e = creatureb.distance(nearest_enemy) if dist_e < 150: reward -= (150 - dist_e) / 150 * 3 if prev_distance_b is not None: reward += (prev_distance_b - dist) * 0.5 if dist < 20: creaturebscore += 1 reset_round() return +35 prev_distance_b = dist return reward def update_nn_b(state, action, reward, next_state): q_vals, _ = dqnb.forward(state) next_q, _ = dqnb.forward(next_state) target = q_vals.copy() target[action] = reward + 0.9 * np.max(next_q) target = np.clip(target, -50, 50) dqnb.train(state, target) def wall_proximity(x, y): # Distance to outer wall d_outer = TRACK_OUTER - max(abs(x), abs(y)) # Distance to inner wall d_inner = max(abs(x), abs(y)) - TRACK_INNER return min(d_outer, d_inner) #aborted system that was used once for active simontaneous tracking of all datas the agents get... #sample: def bruh(): pass print("""{(('R', 'D', 'EU', 'NEAR', None), 'up'): 0, (('R', 'D', 'EU', 'NEAR', 'down'), 'up'): -1.0, (('R', 'D', 'EU', 'NEAR', None), 'down'): -0.20596867247512396, (('R', 'D', 'EU', 'NEAR', 'down'), 'down'): -0.5711609068717072, (('R', 'D', 'EU', 'NEAR', None), 'left'): 0, (('R', 'D', 'EU', 'NEAR', 'down'), 'left'): -0.32252000739479125, (('R', 'D', 'EU', 'NEAR', None), 'right'): 0, (('R', 'D', 'EU', 'NEAR', 'down'), 'right'): 0.38244032440635906, (('R', 'D', 'ER', 'NEAR', 'down'), 'up'): 0, (('R', 'D', 'ER', 'NEAR', 'down'), 'down'): 0, (('R', 'D', 'ER', 'NEAR', 'down'), 'left'): 0, (('R', 'D', 'ER', 'NEAR', 'down'), 'right'): -1.0, (('R', 'D', 'ER', 'NEAR', 'up'), 'up'): 0, (('R', 'D', 'ER', 'NEAR', 'up'), 'down'): 0.19515537688340223, (('R', 'D', 'ER', 'NEAR', 'up'), 'left'): 0, (('R', 'D', 'ER', 'NEAR', 'up'), 'right'): 0, (('R', 'D', 'EU', 'NEAR', 'up'), 'up'): -1.8, (('R', 'D', 'EU', 'NEAR', 'up'), 'down'): 0.194031327524876, (('R', 'D', 'EU', 'NEAR', 'up'), 'left'): -0.5778900000000001, (('R', 'D', 'EU', 'NEAR', 'up'), 'right'): 0, (('R', 'D', 'ER', 'NEAR', 'left'), 'up'): -1.6452122435236889, (('R', 'D', 'ER', 'NEAR', 'left'), 'down'): 0.40742661080044495, (('R', 'D', 'ER', 'NEAR', 'left'), 'left'): -0.36872032160987595, (('R', 'D', 'ER', 'NEAR', 'left'), 'right'): 0.1925, (('R', 'D', 'ED', 'NEAR', 'right'), 'up'): 0, (('R', 'D', 'ED', 'NEAR', 'right'), 'down'): -1.8, (('R', 'D', 'ED', 'NEAR', 'right'), 'left'): 0, (('R', 'D', 'ED', 'NEAR', 'right'), 'right'): 0.19090371317883503, (('R', 'D', 'EL', 'NEAR', 'down'), 'up'): 0, (('R', 'D', 'EL', 'NEAR', 'down'), 'down'): 0, (('R', 'D', 'EL', 'NEAR', 'down'), 'left'): 0, (('R', 'D', 'EL', 'NEAR', 'down'), 'right'): 0, (('R', 'D', 'ED', 'NEAR', 'down'), 'up'): 0, (('R', 'D', 'ED', 'NEAR', 'down'), 'down'): 0, (('R', 'D', 'ED', 'NEAR', 'down'), 'left'): 0, (('R', 'D', 'ED', 'NEAR', 'down'), 'right'): 0, (('R', 'D', 'ER', 'NEAR', 'right'), 'up'): -0.9648720321609876, (('R', 'D', 'ER', 'NEAR', 'right'), 'down'): 0, (('R', 'D', 'ER', 'NEAR', 'right'), 'left'): 0, (('R', 'D', 'ER', 'NEAR', 'right'), 'right'): 0, (('R', 'D', 'EU', 'NEAR', 'right'), 'up'): -0.9648720321609876, (('R', 'D', 'EU', 'NEAR', 'right'), 'down'): 0, (('R', 'D', 'EU', 'NEAR', 'right'), 'left'): 0, (('R', 'D', 'EU', 'NEAR', 'right'), 'right'): 0, (('R', 'D', 'ER', 'MID', 'down'), 'up'): 0, (('R', 'D', 'ER', 'MID', 'down'), 'down'): 0, (('R', 'D', 'ER', 'MID', 'down'), 'left'): -0.32645341813671325, (('R', 'D', 'ER', 'MID', 'down'), 'right'): 0.19860503731036735, (('R', 'D', 'ER', 'MID', 'left'), 'up'): -0.2, (('R', 'D', 'ER', 'MID', 'left'), 'down'): 0.19860811490261215, (('R', 'D', 'ER', 'MID', 'left'), 'left'): -0.1642505393175298, (('R', 'D', 'ER', 'MID', 'left'), 'right'): 0.19860811490261215, (('R', 'D', 'ER', 'MID', 'right'), 'up'): -0.36220287881918345, (('R', 'D', 'ER', 'MID', 'right'), 'down'): 0, (('R', 'D', 'ER', 'MID', 'right'), 'left'): 0, (('R', 'D', 'ER', 'MID', 'right'), 'right'): 0, (('R', 'D', 'ER', 'MID', 'up'), 'up'): -0.2, (('R', 'D', 'ER', 'MID', 'up'), 'down'): 0.2, (('R', 'D', 'ER', 'MID', 'up'), 'left'): -0.33140043145402387, (('R', 'D', 'ER', 'MID', 'up'), 'right'): 0, (('R', 'D', 'SAFE', 'FAR', 'left'), 'up'): -0.2, (('R', 'D', 'SAFE', 'FAR', 'left'), 'down'): 0, (('R', 'D', 'SAFE', 'FAR', 'left'), 'left'): 0, (('R', 'D', 'SAFE', 'FAR', 'left'), 'right'): 0, (('R', 'D', 'SAFE', 'FAR', 'up'), 'up'): 0, (('R', 'D', 'SAFE', 'FAR', 'up'), 'down'): 0, (('R', 'D', 'SAFE', 'FAR', 'up'), 'left'): 0, (('R', 'D', 'SAFE', 'FAR', 'up'), 'right'): 0.2}""") print("""1 0.99 2 0.99 1 0.99 2 0.99 1 0.99 2 0.99 1 0.98995 2 0.99 1 0.98995 2 0.98995 1 0.9899 2 0.98995 1 0.98985 2 0.9899 1 0.98985 2 0.98985 1 0.9898 2 0.98985 1 0.98975 2 0.98985 1 0.9897 2 0.9898 1 0.98965 2 0.9898 1 0.9896 2 0.9898 1 0.9896 2 0.98975 1 0.98955 2 0.9897 1 0.98955 2 0.9897 1 0.9895 2 0.98965 1 0.9895 2 0.9896 1 0.98945 2 0.98955 1 0.9894000000000001 2 0.98955 1 0.9894000000000001 2 0.9895 1 0.9894000000000001 2 0.9895 1 0.9893500000000001 2 0.9895 1 0.9893000000000001 2 0.9895 1 0.9893000000000001 2 0.98945 1 0.9892500000000001 2 0.9894000000000001 1 0.9892500000000001 2 0.9894000000000001 1 0.9892500000000001 2 0.9894000000000001 1 0.9892000000000001 2 0.9893500000000001 1 0.9892000000000001 2 0.9893000000000001 1 0.9891500000000001 2 0.9892500000000001 1 0.9891500000000001 2 0.9892500000000001 1 0.9891000000000001 2 0.9892500000000001 1 0.9891000000000001 2 0.9892000000000001 1 0.9891000000000001 2 0.9891500000000001 1 0.9890500000000001 2 0.9891000000000001 1 0.9890500000000001 2 0.9891000000000001 1 0.9890000000000001 2 0.9891000000000001 1 0.9889500000000001 2 0.9890500000000001 1 0.9889500000000001 2 0.9890000000000001 1 0.9889500000000001 2 0.9890000000000001""") def mos(): pass print(f"""{state}, {reward}, {next_state}, {epsilon}""") print(f"""{bx},{by},{prev_distance}""") print(f"action: {action}") print(f"last action: {lastaction}") print(f"lastlastaction: {lastlastaction}") #abandoned def move_player(): pass if keys_pressed["w"]: food.sety(min(food.ycor() + 8, TOP_BOUND)) if keys_pressed["s"]: food.sety(max(food.ycor() - 8, BOTTOM_BOUND)) if keys_pressed["a"]: food.setx(max(food.xcor() - 8, LEFT_BOUND)) if keys_pressed["d"]: food.setx(min(food.xcor() + 8, RIGHT_BOUND)) #2nd main function basically. Screen loop cant exist multiple times. So I add everything into this func and call every couple frames. Loop them def update(): global epsilon, prev_distance, lastaction, lastlastaction, epsilon_b, prev_distance2, lastaction2, lastlastaction2, collideok, collideokb, creaturescore, creaturebscore, GAME_OVER if GAME_OVER: return for p in enemies: p.setheading(random.randint(0,360)) p.forward(1) if abs(p.xcor()) > 290 or abs(p.ycor()) > 290: p.goto(random.randint(-250,250),random.randint(-250,250)) if not GAME_OVER: us.clear() us.goto(0,400) us.write(f"AI racing system, Agent A score = {creaturescore}, B score = {creaturebscore}", align="center", font=("Arial", 50, "bold")) state = get_state_numeric() stateb = get_state_numeric_b() action = choose_action_nn(state) actionb = choose_action_nn_b(stateb) do_action(actions[action]) do_action_b(actions_b[actionb]) lastaction = actions[action] lastaction_b = actions_b[actionb] ec = enemy_collision() ecb = enemy_collision_b() reward = get_reward(ec,action) rewardb = get_reward_b(ecb,actionb) if creaturescore >= WIN_SCORE or creaturebscore >= WIN_SCORE: GAME_OVER = True us.clear() us.goto(0, 0) if creaturescore > creaturebscore: us.write("AGENT A WINS 🏆", align="center", font=("Arial", 40, "bold")) else: us.write("AGENT B WINS 🏆", align="center", font=("Arial", 40, "bold")) screen.update() return if track_collision(creature): reward = -100 prev_distance = None reset_agents(creature) q_vals, _ = dqn.forward(state) q_vals[action] = reward dqn.train(state, q_vals) screen.update() screen.ontimer(update, 10) return if track_collision(creatureb): rewardb = -100 prev_distance_b = None reset_agentsb(creatureb) q_vals, _ = dqnb.forward(stateb) q_vals[actionb] = rewardb dqnb.train(stateb, q_vals) screen.update() screen.ontimer(update, 10) return if epsilon < 0.5: collideok = True if epsilon_b < 0.5: collideokb = True if ec or abs(creature.xcor()) > 290 or abs(creature.ycor()) > 290: reward = -100 q_vals, _ = dqn.forward(state) q_vals[action] = reward dqn.train(state, q_vals) prev_distance = None creature.setheading(random.choice([0, 90, 180, 270])) screen.update() screen.ontimer(update, 10) return if ecb or abs(creatureb.xcor()) > 290 or abs(creatureb.ycor()) > 290: rewardb = -100 q_vals, _ = dqnb.forward(stateb) q_vals[actionb] = rewardb dqnb.train(stateb, q_vals) prev_distance_b = None creatureb.setheading(random.choice([0, 90, 180, 270])) screen.update() screen.ontimer(update, 10) return next_state = get_state_numeric() next_state_b = get_state_numeric_b() update_nn(state, action, reward, next_state) update_nn_b(stateb, actionb, rewardb, next_state_b) for p in enemies: if p.xcor() == 0 and p.ycor() == 0: p.goto(random.randint(-250,250), random.randint(-250,250)) epsilon *= 0.995 epsilon = max(epsilon, 0.05) epsilon_b *= 0.995 epsilon_b = max(epsilon_b, 0.05) screen.update() screen.ontimer(update, 10) #abandoned player movement assisting funcs def key_press(key): pass keys_pressed[key] = True def key_release(key): pass keys_pressed[key] = False def advanced_learning(): pass epsilon *= 0.999971 epsilon = max(epsilon, 0.05) epsilon_b *= 0.999971 epsilon_b = max(epsilon_b, 0.05) screen.listen() screen.onkeypress(lambda: key_press("w"), "w") screen.onkeyrelease(lambda: key_release("w"), "w") screen.onkeypress(lambda: key_press("s"), "s") screen.onkeyrelease(lambda: key_release("s"), "s") screen.onkeypress(lambda: key_press("a"), "a") screen.onkeyrelease(lambda: key_release("a"), "a") screen.onkeypress(lambda: key_press("d"), "d") screen.onkeyrelease(lambda: key_release("d"), "d") def main(): minesotapineapple() reset_agents(creature) reset_agentsb(creatureb) placefoodontrack(food) drawtrack() update() show_start_screen(main) turtle.mainloop()