Polymorphic brain
Paste ID: a0ad09b0
Created at: 2026-09-25 05:13:48
Content
"""
PGBA - Polymorphic Genetic Basin AI
-----------------------------------
A computational prototype of:
Genetic evolution
+
Polymorphic neural/decision topology
+
Reward heat map
+
Danger heat map
+
Basin-of-attraction dynamics
+
Future prediction
+
Reinforcement learning
Core equation:
P(x,t) = R(x,t) - lambda * D(x,t)
Decision:
a = Attractor[
gradient(
R - lambda*D + gamma*FuturePrediction
)
]
The implementation deliberately uses only Python's standard library.
"""
from dataclasses import dataclass, field
import math
import random
import copy
import time
# ============================================================
# BASIC UTILITIES
# ============================================================
ACTIONS = ["UP", "DOWN", "LEFT", "RIGHT", "EAT", "WAIT"]
DIRECTION = {
"UP": (0, -1),
"DOWN": (0, 1),
"LEFT": (-1, 0),
"RIGHT": (1, 0),
"EAT": (0, 0),
"WAIT": (0, 0),
}
def clamp(value, low, high):
return max(low, min(high, value))
def softmax(values, temperature=1.0):
temperature = max(temperature, 0.001)
m = max(values)
exps = [math.exp((v - m) / temperature) for v in values]
total = sum(exps)
return [x / total for x in exps]
def distance(a, b):
return math.sqrt(
(a[0] - b[0]) ** 2 +
(a[1] - b[1]) ** 2
)
def gaussian(d, sigma):
return math.exp(
-(d * d) / (2.0 * sigma * sigma)
)
# ============================================================
# GENOME
# ============================================================
@dataclass
class Genome:
# Heat-map parameters
food_drive: float = 1.0
danger_sensitivity: float = 1.0
shelter_drive: float = 0.6
curiosity: float = 0.2
uncertainty_aversion: float = 0.2
# Dynamics
danger_weight: float = 1.5
discount: float = 0.90
learning_rate: float = 0.08
# Prediction
prediction_horizon: int = 3
# Plasticity
growth_threshold: float = 3.0
prune_threshold: float = 0.05
# Exploration
temperature: float = 0.35
# Evolution
mutation_rate: float = 0.12
def mutate(self):
fields = [
"food_drive",
"danger_sensitivity",
"shelter_drive",
"curiosity",
"uncertainty_aversion",
"danger_weight",
"discount",
"learning_rate",
"prediction_horizon",
"growth_threshold",
"prune_threshold",
"temperature",
]
for name in fields:
if random.random() < self.mutation_rate:
value = getattr(self, name)
if name == "prediction_horizon":
value += random.choice([-1, 1])
value = int(clamp(value, 1, 8))
else:
value += random.gauss(0, 0.15 * max(abs(value), 0.3))
setattr(self, name, value)
self.food_drive = clamp(self.food_drive, 0.05, 4.0)
self.danger_sensitivity = clamp(
self.danger_sensitivity, 0.05, 5.0
)
self.shelter_drive = clamp(
self.shelter_drive, 0.0, 4.0
)
self.curiosity = clamp(
self.curiosity, 0.0, 3.0
)
self.uncertainty_aversion = clamp(
self.uncertainty_aversion, 0.0, 3.0
)
self.danger_weight = clamp(
self.danger_weight, 0.1, 5.0
)
self.discount = clamp(
self.discount, 0.5, 0.99
)
self.learning_rate = clamp(
self.learning_rate, 0.005, 0.5
)
self.growth_threshold = clamp(
self.growth_threshold, 0.5, 10.0
)
self.prune_threshold = clamp(
self.prune_threshold, 0.001, 1.0
)
self.temperature = clamp(
self.temperature, 0.03, 2.0
)
def crossover(a, b):
child = Genome()
for name in vars(child):
if random.random() < 0.5:
setattr(child, name, copy.deepcopy(getattr(a, name)))
else:
setattr(child, name, copy.deepcopy(getattr(b, name)))
child.mutate()
return child
# ============================================================
# POLYMORPHIC BRAIN
# ============================================================
@dataclass
class BrainNode:
name: str
utility: float = 1.0
age: int = 0
class PolymorphicBrain:
def __init__(self):
self.nodes = {}
self.edges = {}
self.activation = {}
self.create_node("vision")
self.create_node("reward_field")
self.create_node("danger_field")
self.create_node("basin")
self.create_node("prediction")
self.create_node("decision")
self.connect("vision", "reward_field")
self.connect("vision", "danger_field")
self.connect("reward_field", "basin")
self.connect("danger_field", "basin")
self.connect("basin", "prediction")
self.connect("prediction", "decision")
# --------------------------------------------------------
def create_node(self, name):
if name not in self.nodes:
self.nodes[name] = BrainNode(name)
self.edges[name] = []
self.activation[name] = 0.0
# --------------------------------------------------------
def connect(self, a, b):
self.create_node(a)
self.create_node(b)
if b not in self.edges[a]:
self.edges[a].append(b)
# --------------------------------------------------------
def remove_node(self, name):
if name in self.nodes:
del self.nodes[name]
if name in self.edges:
del self.edges[name]
for edges in self.edges.values():
if name in edges:
edges.remove(name)
if name in self.activation:
del self.activation[name]
# --------------------------------------------------------
def activate(self, name, value):
if name in self.activation:
self.activation[name] = value
self.nodes[name].utility *= 0.99
self.nodes[name].utility += abs(value) * 0.01
# --------------------------------------------------------
def adapt(self, prediction_error, danger_signal):
"""
Polymorphic structural adaptation.
Large prediction errors cause the system to grow
specialized modules.
Repeatedly useless modules can later be pruned.
"""
if prediction_error > 2.5:
if danger_signal > 0.5:
self.create_node("threat_memory")
self.connect(
"danger_field",
"threat_memory"
)
self.connect(
"threat_memory",
"prediction"
)
else:
self.create_node("resource_memory")
self.connect(
"reward_field",
"resource_memory"
)
self.connect(
"resource_memory",
"prediction"
)
if prediction_error > 4.0:
self.create_node("counterfactual")
self.connect(
"prediction",
"counterfactual"
)
self.connect(
"counterfactual",
"decision"
)
# Age all modules
for node in self.nodes.values():
node.age += 1
# Increase utility of active pathways
for node in self.nodes.values():
act = abs(self.activation.get(node.name, 0))
node.utility += act * 0.02
node.utility *= 0.999
# --------------------------------------------------------
def prune(self, threshold):
protected = {
"vision",
"reward_field",
"danger_field",
"basin",
"prediction",
"decision",
}
remove = []
for name, node in self.nodes.items():
if name not in protected:
if node.utility < threshold and node.age > 100:
remove.append(name)
for name in remove:
self.remove_node(name)
# --------------------------------------------------------
def description(self):
lines = []
lines.append(
f"Brain nodes: {len(self.nodes)}"
)
lines.append(
f"Brain connections: "
f"{sum(len(x) for x in self.edges.values())}"
)
lines.append(
"Topology:"
)
for name, targets in self.edges.items():
if targets:
lines.append(
f" {name} -> {', '.join(targets)}"
)
return "\n".join(lines)
# ============================================================
# WORLD
# ============================================================
class World:
def __init__(self, size=15):
self.size = size
self.reset()
# --------------------------------------------------------
def reset(self):
self.agent = (
self.size // 2,
self.size // 2
)
self.enemy = self.random_position(
minimum_distance=5
)
self.shelter = (
1,
1
)
self.food = set()
for _ in range(10):
self.food.add(
self.random_position()
)
self.energy = 50.0
self.score = 0.0
self.time = 0
self.dead = False
self.collected = 0
# --------------------------------------------------------
def random_position(self, minimum_distance=0):
while True:
p = (
random.randrange(self.size),
random.randrange(self.size)
)
if distance(p, self.agent) >= minimum_distance:
return p
# --------------------------------------------------------
def inside(self, p):
x, y = p
return (
0 <= x < self.size and
0 <= y < self.size
)
# --------------------------------------------------------
def move_enemy(self):
ex, ey = self.enemy
ax, ay = self.agent
dx = 0
dy = 0
if ax > ex:
dx = 1
elif ax < ex:
dx = -1
if ay > ey:
dy = 1
elif ay < ey:
dy = -1
# Sometimes predator wanders
if random.random() < 0.15:
dx, dy = random.choice([
(0, 0),
(1, 0),
(-1, 0),
(0, 1),
(0, -1)
])
new_pos = (
ex + dx,
ey + dy
)
if self.inside(new_pos):
self.enemy = new_pos
# --------------------------------------------------------
def step(self, action):
reward = -0.03
old_position = self.agent
dx, dy = DIRECTION[action]
new_position = (
old_position[0] + dx,
old_position[1] + dy
)
if self.inside(new_position):
self.agent = new_position
else:
reward -= 1.0
# ----------------------------------------------------
# Eat
# ----------------------------------------------------
if action == "EAT":
if self.agent in self.food:
self.food.remove(self.agent)
reward += 5.0
self.energy += 20
self.collected += 1
else:
reward -= 0.2
# ----------------------------------------------------
# Shelter
# ----------------------------------------------------
if self.agent == self.shelter:
reward += 0.15
self.energy += 0.5
# ----------------------------------------------------
# Predator
# ----------------------------------------------------
self.move_enemy()
if distance(self.agent, self.enemy) <= 1.0:
reward -= 8.0
self.energy -= 15
# ----------------------------------------------------
# Energy
# ----------------------------------------------------
self.energy -= 0.25
if self.energy <= 0:
reward -= 10
self.dead = True
self.score += reward
self.time += 1
if self.time >= 250:
self.dead = True
return reward
# ============================================================
# HEAT MAP SYSTEM
# ============================================================
class HeatMaps:
def __init__(self, world, genome):
self.world = world
self.genome = genome
self.reward = self.empty_map()
self.danger = self.empty_map()
self.potential = self.empty_map()
self.future = self.empty_map()
# --------------------------------------------------------
def empty_map(self):
return [
[0.0 for _ in range(self.world.size)]
for _ in range(self.world.size)
]
# --------------------------------------------------------
def build(self):
self.build_reward()
self.build_danger()
self.combine()
# --------------------------------------------------------
def build_reward(self):
w = self.world
for y in range(w.size):
for x in range(w.size):
p = (x, y)
value = 0.0
# Food attraction
for food in w.food:
d = distance(p, food)
value += (
self.genome.food_drive
* gaussian(d, 2.5)
* 5.0
)
# Shelter attraction
d = distance(p, w.shelter)
shelter_strength = 0.4
if w.energy < 20:
shelter_strength = (
self.genome.shelter_drive
* 2.0
)
value += (
shelter_strength
* gaussian(d, 5.0)
)
# Curiosity rewards distant/unknown locations
center = w.size / 2
edge_distance = min(
x,
y,
w.size - 1 - x,
w.size - 1 - y
)
exploration = (
(1.0 - edge_distance / center)
* self.genome.curiosity
)
value += exploration
self.reward[y][x] = value
# --------------------------------------------------------
def build_danger(self):
w = self.world
for y in range(w.size):
for x in range(w.size):
p = (x, y)
d = distance(p, w.enemy)
danger = (
self.genome.danger_sensitivity
* gaussian(d, 2.7)
* 8.0
)
# Predict predator movement toward the agent.
ex, ey = w.enemy
ax, ay = w.agent
px = ex + (
1 if ax > ex else
-1 if ax < ex else
0
)
py = ey + (
1 if ay > ey else
-1 if ay < ey else
0
)
predicted_enemy = (px, py)
predicted_distance = distance(
p,
predicted_enemy
)
danger += (
self.genome.danger_sensitivity
* gaussian(
predicted_distance,
3.0
)
* 5.0
)
# Boundary uncertainty
edge = min(
x,
y,
w.size - 1 - x,
w.size - 1 - y
)
if edge == 0:
danger += (
self.genome.uncertainty_aversion
)
self.danger[y][x] = danger
# --------------------------------------------------------
def combine(self):
for y in range(self.world.size):
for x in range(self.world.size):
self.potential[y][x] = (
self.reward[y][x]
-
self.genome.danger_weight
* self.danger[y][x]
)
# ============================================================
# BASIN MATHEMATICS
# ============================================================
class BasinSolver:
def __init__(self, maps):
self.maps = maps
# --------------------------------------------------------
def neighbors(self, p):
x, y = p
result = []
for action in [
"UP",
"DOWN",
"LEFT",
"RIGHT"
]:
dx, dy = DIRECTION[action]
q = (
x + dx,
y + dy
)
if (
0 <= q[0] < self.maps.world.size
and
0 <= q[1] < self.maps.world.size
):
result.append((action, q))
return result
# --------------------------------------------------------
def gradient(self, p):
x, y = p
size = self.maps.world.size
left = self.maps.potential[y][
max(0, x - 1)
]
right = self.maps.potential[y][
min(size - 1, x + 1)
]
up = self.maps.potential[
max(0, y - 1)
][x]
down = self.maps.potential[
min(size - 1, y + 1)
][x]
gx = (right - left) / 2.0
gy = (down - up) / 2.0
return gx, gy
# --------------------------------------------------------
def trace_basin(self, start, max_steps=30):
current = start
visited = set()
for _ in range(max_steps):
if current in visited:
break
visited.add(current)
neighbors = self.neighbors(current)
if not neighbors:
break
best = max(
neighbors,
key=lambda item:
self.maps.potential[
item[1][1]
][
item[1][0]
]
)
current_value = self.maps.potential[
current[1]
][
current[0]
]
best_value = self.maps.potential[
best[1][1]
][
best[1][0]
]
if best_value <= current_value:
break
current = best[1]
return current
# --------------------------------------------------------
def first_basin_action(self, start):
target = self.trace_basin(start)
if target == start:
return "WAIT"
x, y = start
tx, ty = target
if tx > x:
return "RIGHT"
if tx < x:
return "LEFT"
if ty > y:
return "DOWN"
if ty < y:
return "UP"
return "WAIT"
# ============================================================
# AGENT
# ============================================================
class Agent:
def __init__(self, genome=None):
self.genome = genome or Genome()
self.brain = PolymorphicBrain()
# Learned action values
self.q = {
action: 0.0
for action in ACTIONS
}
# Lifetime plasticity
self.plastic_food = 0.0
self.plastic_danger = 0.0
self.last_action = None
self.last_prediction = 0.0
self.total_error = 0.0
self.decisions = 0
# --------------------------------------------------------
def effective_food_drive(self):
return max(
0.01,
self.genome.food_drive
+ self.plastic_food
)
# --------------------------------------------------------
def effective_danger_sensitivity(self):
return max(
0.01,
self.genome.danger_sensitivity
+ self.plastic_danger
)
# --------------------------------------------------------
def make_maps(self, world):
genome = copy.deepcopy(self.genome)
genome.food_drive = self.effective_food_drive()
genome.danger_sensitivity = (
self.effective_danger_sensitivity()
)
maps = HeatMaps(
world,
genome
)
maps.build()
return maps
# --------------------------------------------------------
def predict_future_value(
self,
world,
position,
action
):
"""
Short-horizon prediction.
The agent estimates the potential after taking
an action, then looks ahead through the local
potential landscape.
"""
dx, dy = DIRECTION[action]
next_position = (
position[0] + dx,
position[1] + dy
)
if not world.inside(next_position):
return -10.0
maps = self.make_maps(world)
x, y = next_position
immediate = maps.potential[y][x]
best_future = immediate
current = next_position
for _ in range(
self.genome.prediction_horizon
):
candidates = []
cx, cy = current
for direction in [
"UP",
"DOWN",
"LEFT",
"RIGHT"
]:
ddx, ddy = DIRECTION[direction]
q = (
cx + ddx,
cy + ddy
)
if world.inside(q):
candidates.append(
maps.potential[q[1]][q[0]]
)
if not candidates:
break
best_future = max(
best_future,
max(candidates)
)
# Follow best predicted basin
best_index = max(
range(len(candidates)),
key=lambda i: candidates[i]
)
dirs = [
"UP",
"DOWN",
"LEFT",
"RIGHT"
]
ddx, ddy = DIRECTION[
dirs[best_index]
]
proposed = (
current[0] + ddx,
current[1] + ddy
)
if world.inside(proposed):
current = proposed
return (
immediate
+
self.genome.discount
* best_future
)
# --------------------------------------------------------
def choose_action(self, world):
maps = self.make_maps(world)
solver = BasinSolver(maps)
position = world.agent
gx, gy = solver.gradient(position)
basin_action = solver.first_basin_action(
position
)
self.brain.activate(
"vision",
1.0
)
self.brain.activate(
"reward_field",
maps.reward[position[1]][position[0]]
)
self.brain.activate(
"danger_field",
maps.danger[position[1]][position[0]]
)
self.brain.activate(
"basin",
math.sqrt(gx * gx + gy * gy)
)
action_scores = []
for action in ACTIONS:
future = self.predict_future_value(
world,
position,
action
)
learned = self.q[action]
score = (
future
+
learned
)
# Fast basin signal
if action == basin_action:
score += 1.5
# Eating is only useful where food exists
if action == "EAT":
if position in world.food:
score += 5.0
else:
score -= 1.0
action_scores.append(score)
probabilities = softmax(
action_scores,
self.genome.temperature
)
action = random.choices(
ACTIONS,
weights=probabilities
)[0]
# Slightly reduce randomness once the basin
# direction is extremely clear.
if random.random() < 0.15:
action = basin_action
self.last_action = action
self.last_prediction = max(
action_scores
)
self.decisions += 1
return action
# --------------------------------------------------------
def learn(
self,
reward,
next_world,
old_world
):
if self.last_action is None:
return
next_maps = self.make_maps(
next_world
)
x, y = next_world.agent
future_value = max(
next_maps.potential[y][x],
max(self.q.values())
)
prediction = self.q[
self.last_action
]
target = (
reward
+
self.genome.discount
* future_value
)
error = target - prediction
self.q[
self.last_action
] += (
self.genome.learning_rate
* error
)
self.total_error += abs(error)
# Plasticity changes the underlying
# behavioral parameters during life.
if reward > 1:
self.plastic_food += (
0.002 * reward
)
if reward < -1:
self.plastic_danger += (
0.003 * abs(reward)
)
self.plastic_food = clamp(
self.plastic_food,
-1.0,
1.5
)
self.plastic_danger = clamp(
self.plastic_danger,
-1.0,
2.0
)
maps = self.make_maps(
old_world
)
danger_signal = maps.danger[
old_world.agent[1]
][
old_world.agent[0]
]
self.brain.adapt(
abs(error),
danger_signal
)
self.brain.prune(
self.genome.prune_threshold
)
# --------------------------------------------------------
def explain(self, world):
maps = self.make_maps(world)
position = world.agent
solver = BasinSolver(maps)
gx, gy = solver.gradient(position)
basin = solver.trace_basin(
position
)
return f"""
CURRENT AGENT STATE
-------------------
Position: {position}
Energy: {world.energy:.2f}
Score: {world.score:.2f}
Food collected: {world.collected}
FIELD STATE
-----------
Reward R(x,y):
{maps.reward[position[1]][position[0]]:.3f}
Danger D(x,y):
{maps.danger[position[1]][position[0]]:.3f}
Combined potential:
P = R - lambda*D
P(position):
{maps.potential[position[1]][position[0]]:.3f}
Gradient:
gx = {gx:.3f}
gy = {gy:.3f}
Current basin attractor:
{basin}
Brain nodes:
{len(self.brain.nodes)}
Brain connections:
{sum(len(x) for x in self.brain.edges.values())}
"""
# --------------------------------------------------------
def answer(self, question, world):
q = question.lower()
if "brain" in q:
return self.brain.description()
if "genome" in q:
return (
"Genome:\n"
+
"\n".join(
f"{k}: {v}"
for k, v in vars(
self.genome
).items()
)
)
if "reward" in q:
maps = self.make_maps(world)
return (
"The reward heat map represents "
"estimated desirability at every "
"location. Food creates positive "
"Gaussian fields and shelter becomes "
"more valuable when energy is low.\n\n"
f"Current reward value: "
f"{maps.reward[world.agent[1]][world.agent[0]]:.3f}"
)
if "danger" in q or "threat" in q:
maps = self.make_maps(world)
return (
"The danger heat map estimates future "
"risk at every location, including the "
"predicted predator position.\n\n"
f"Current danger value: "
f"{maps.danger[world.agent[1]][world.agent[0]]:.3f}"
)
if "basin" in q:
maps = self.make_maps(world)
solver = BasinSolver(maps)
target = solver.trace_basin(
world.agent
)
return (
"A basin is a region of the potential "
"landscape whose local gradient leads "
"toward the same attractor.\n\n"
f"Current attractor: {target}\n"
f"First basin action: "
f"{solver.first_basin_action(world.agent)}"
)
if "flee" in q or "enemy" in q:
return (
"I increase the danger contribution "
"D(x,y), subtract it from reward through "
"P = R - lambda*D, and therefore move "
"toward regions whose combined potential "
"is safer."
)
if "food" in q:
return (
"Food creates positive reward basins. "
"The food-drive genome controls their "
"strength. The agent can also learn a "
"plastic increase in food attraction."
)
if "learn" in q or "training" in q:
return (
f"Decisions: {self.decisions}\n"
f"Accumulated prediction error: "
f"{self.total_error:.3f}\n"
f"Plastic food adjustment: "
f"{self.plastic_food:.3f}\n"
f"Plastic danger adjustment: "
f"{self.plastic_danger:.3f}"
)
if "what do you know" in q or "state" in q:
return self.explain(world)
return (
"I understand questions about my "
"brain, genome, reward map, danger map, "
"basins, food, enemies, learning, training, "
"and current state."
)
# ============================================================
# EVOLUTION
# ============================================================
class Evolution:
def __init__(
self,
population_size=20
):
self.population_size = population_size
self.population = [
Agent()
for _ in range(population_size)
]
self.generation = 0
# --------------------------------------------------------
def episode(self, agent):
world = World()
while not world.dead:
old_world = copy.deepcopy(world)
action = agent.choose_action(
world
)
reward = world.step(
action
)
agent.learn(
reward,
world,
old_world
)
fitness = (
world.score
+
world.collected * 2.0
+
world.time * 0.03
)
return fitness
# --------------------------------------------------------
def train_generation(self):
results = []
for agent in self.population:
# Reset lifetime plasticity between generations
agent.plastic_food = 0
agent.plastic_danger = 0
fitness = self.episode(
agent
)
results.append(
(fitness, agent)
)
results.sort(
key=lambda x: x[0],
reverse=True
)
survivors = results[
:max(
2,
self.population_size // 4
)
]
new_population = [
copy.deepcopy(agent)
for _, agent in survivors
]
while len(new_population) < self.population_size:
_, parent_a = random.choice(
survivors
)
_, parent_b = random.choice(
survivors
)
genome = crossover(
parent_a.genome,
parent_b.genome
)
child = Agent(genome)
new_population.append(
child
)
self.population = new_population
self.generation += 1
return results
# --------------------------------------------------------
def train(self, generations=20):
print(
f"\nTraining {generations} generations..."
)
for _ in range(generations):
results = self.train_generation()
fitness_values = [
x[0]
for x in results
]
mean_fitness = (
sum(fitness_values)
/
len(fitness_values)
)
best_fitness = max(
fitness_values
)
best_agent = results[0][1]
print(
f"Generation "
f"{self.generation:03d} | "
f"mean={mean_fitness:7.2f} | "
f"best={best_fitness:7.2f} | "
f"brain={len(best_agent.brain.nodes)} "
f"nodes"
)
print("\nTraining complete.")
# --------------------------------------------------------
def best_agent(self):
return self.population[0]
# ============================================================
# ASCII VISUALIZATION
# ============================================================
def print_world(world):
print("\nWORLD")
for y in range(world.size):
row = ""
for x in range(world.size):
p = (x, y)
if p == world.agent:
char = "A"
elif p == world.enemy:
char = "X"
elif p == world.shelter:
char = "S"
elif p in world.food:
char = "*"
else:
char = "."
row += char + " "
print(row)
def print_heatmap(grid, world, title):
print(f"\n{title}")
maximum = max(
max(row)
for row in grid
)
minimum = min(
min(row)
for row in grid
)
chars = " .:-=+*#%@"
for y in range(world.size):
row = ""
for x in range(world.size):
value = grid[y][x]
if maximum == minimum:
index = 0
else:
normalized = (
(value - minimum)
/
(maximum - minimum)
)
index = int(
normalized
* (len(chars) - 1)
)
row += chars[index]
print(row)
# ============================================================
# TEST TRAINED AGENT
# ============================================================
def test_agent(agent):
world = World()
print(
"\nRunning trained agent..."
)
for step in range(100):
if world.dead:
break
action = agent.choose_action(
world
)
reward = world.step(
action
)
print(
f"step={step:03d} "
f"pos={world.agent} "
f"action={action:5s} "
f"reward={reward:6.2f} "
f"energy={world.energy:6.2f}"
)
time.sleep(0.01)
print(
"\nFinal score:",
round(world.score, 2)
)
return world
# ============================================================
# INTERACTIVE SHELL
# ============================================================
def interactive(agent):
world = World()
print(
"""
============================================================
PGBA INTERACTIVE AGENT
============================================================
Commands:
train N
test
world
reward
danger
potential
brain
genome
state
ask <question>
step
help
quit
Examples:
ask why did you flee?
ask describe your brain
ask what is my reward map?
ask what is a basin?
ask what have you learned?
"""
)
while True:
try:
command = input(
"\nPGBA> "
).strip()
except EOFError:
break
if not command:
continue
lower = command.lower()
# ----------------------------------------------------
if lower == "quit":
break
# ----------------------------------------------------
elif lower == "help":
print(
"Use: train 10, test, world, reward, "
"danger, potential, brain, genome, "
"state, ask <question>, step."
)
# ----------------------------------------------------
elif lower.startswith("train"):
parts = command.split()
amount = 5
if len(parts) > 1:
try:
amount = int(parts[1])
except ValueError:
pass
evolution.train(amount)
agent = evolution.best_agent()
world = World()
# ----------------------------------------------------
elif lower == "test":
world = test_agent(
agent
)
# ----------------------------------------------------
elif lower == "step":
if world.dead:
world = World()
old_world = copy.deepcopy(
world
)
action = agent.choose_action(
world
)
reward = world.step(
action
)
agent.learn(
reward,
world,
old_world
)
print(
f"Action={action} "
f"Reward={reward:.2f} "
f"Position={world.agent} "
f"Energy={world.energy:.2f}"
)
# ----------------------------------------------------
elif lower == "world":
print_world(world)
# ----------------------------------------------------
elif lower == "reward":
maps = agent.make_maps(
world
)
print_heatmap(
maps.reward,
world,
"REWARD HEAT MAP R(x,y)"
)
# ----------------------------------------------------
elif lower == "danger":
maps = agent.make_maps(
world
)
print_heatmap(
maps.danger,
world,
"DANGER HEAT MAP D(x,y)"
)
# ----------------------------------------------------
elif lower == "potential":
maps = agent.make_maps(
world
)
print_heatmap(
maps.potential,
world,
"COMBINED POTENTIAL P = R - lambda*D"
)
# ----------------------------------------------------
elif lower == "brain":
print(
agent.brain.description()
)
# ----------------------------------------------------
elif lower == "genome":
for k, v in vars(
agent.genome
).items():
print(
f"{k:25s} {v}"
)
# ----------------------------------------------------
elif lower == "state":
print(
agent.explain(world)
)
# ----------------------------------------------------
elif lower.startswith("ask "):
question = command[4:]
print(
"\n" +
agent.answer(
question,
world
)
)
# ----------------------------------------------------
else:
print(
agent.answer(
command,
world
)
)
# ============================================================
# MAIN
# ============================================================
if __name__ == "__main__":
random.seed()
print(
"""
============================================================
POLYMORPHIC GENETIC BASIN AI
============================================================
Starting evolutionary training...
"""
)
evolution = Evolution(
population_size=20
)
# Initial training
evolution.train(
generations=20
)
agent = evolution.best_agent()
print(
"\nBest evolved genome:"
)
for key, value in vars(
agent.genome
).items():
print(
f" {key:25s}: {value}"
)
print(
"\nEvolved brain:"
)
print(
agent.brain.description()
)
# Start interaction
interactive(agent)