Polymorphic brain

Paste ID: a0ad09b0

Created at: 2026-09-25 05:13:48

code Python

Content

"""
PGBA - Polymorphic Genetic Basin AI
-----------------------------------

A computational prototype of:

    Genetic evolution
          +
    Polymorphic neural/decision topology
          +
    Reward heat map
          +
    Danger heat map
          +
    Basin-of-attraction dynamics
          +
    Future prediction
          +
    Reinforcement learning

Core equation:

    P(x,t) = R(x,t) - lambda * D(x,t)

Decision:

    a = Attractor[
            gradient(
                R - lambda*D + gamma*FuturePrediction
            )
        ]

The implementation deliberately uses only Python's standard library.
"""

from dataclasses import dataclass, field
import math
import random
import copy
import time


# ============================================================
# BASIC UTILITIES
# ============================================================

ACTIONS = ["UP", "DOWN", "LEFT", "RIGHT", "EAT", "WAIT"]

DIRECTION = {
    "UP": (0, -1),
    "DOWN": (0, 1),
    "LEFT": (-1, 0),
    "RIGHT": (1, 0),
    "EAT": (0, 0),
    "WAIT": (0, 0),
}


def clamp(value, low, high):
    return max(low, min(high, value))


def softmax(values, temperature=1.0):
    temperature = max(temperature, 0.001)

    m = max(values)
    exps = [math.exp((v - m) / temperature) for v in values]
    total = sum(exps)

    return [x / total for x in exps]


def distance(a, b):
    return math.sqrt(
        (a[0] - b[0]) ** 2 +
        (a[1] - b[1]) ** 2
    )


def gaussian(d, sigma):
    return math.exp(
        -(d * d) / (2.0 * sigma * sigma)
    )


# ============================================================
# GENOME
# ============================================================

@dataclass
class Genome:

    # Heat-map parameters
    food_drive: float = 1.0
    danger_sensitivity: float = 1.0
    shelter_drive: float = 0.6
    curiosity: float = 0.2
    uncertainty_aversion: float = 0.2

    # Dynamics
    danger_weight: float = 1.5
    discount: float = 0.90
    learning_rate: float = 0.08

    # Prediction
    prediction_horizon: int = 3

    # Plasticity
    growth_threshold: float = 3.0
    prune_threshold: float = 0.05

    # Exploration
    temperature: float = 0.35

    # Evolution
    mutation_rate: float = 0.12

    def mutate(self):

        fields = [
            "food_drive",
            "danger_sensitivity",
            "shelter_drive",
            "curiosity",
            "uncertainty_aversion",
            "danger_weight",
            "discount",
            "learning_rate",
            "prediction_horizon",
            "growth_threshold",
            "prune_threshold",
            "temperature",
        ]

        for name in fields:

            if random.random() < self.mutation_rate:

                value = getattr(self, name)

                if name == "prediction_horizon":
                    value += random.choice([-1, 1])
                    value = int(clamp(value, 1, 8))

                else:
                    value += random.gauss(0, 0.15 * max(abs(value), 0.3))

                setattr(self, name, value)

        self.food_drive = clamp(self.food_drive, 0.05, 4.0)
        self.danger_sensitivity = clamp(
            self.danger_sensitivity, 0.05, 5.0
        )
        self.shelter_drive = clamp(
            self.shelter_drive, 0.0, 4.0
        )
        self.curiosity = clamp(
            self.curiosity, 0.0, 3.0
        )
        self.uncertainty_aversion = clamp(
            self.uncertainty_aversion, 0.0, 3.0
        )
        self.danger_weight = clamp(
            self.danger_weight, 0.1, 5.0
        )
        self.discount = clamp(
            self.discount, 0.5, 0.99
        )
        self.learning_rate = clamp(
            self.learning_rate, 0.005, 0.5
        )
        self.growth_threshold = clamp(
            self.growth_threshold, 0.5, 10.0
        )
        self.prune_threshold = clamp(
            self.prune_threshold, 0.001, 1.0
        )
        self.temperature = clamp(
            self.temperature, 0.03, 2.0
        )


def crossover(a, b):

    child = Genome()

    for name in vars(child):

        if random.random() < 0.5:
            setattr(child, name, copy.deepcopy(getattr(a, name)))
        else:
            setattr(child, name, copy.deepcopy(getattr(b, name)))

    child.mutate()

    return child


# ============================================================
# POLYMORPHIC BRAIN
# ============================================================

@dataclass
class BrainNode:

    name: str
    utility: float = 1.0
    age: int = 0


class PolymorphicBrain:

    def __init__(self):

        self.nodes = {}

        self.edges = {}

        self.activation = {}

        self.create_node("vision")
        self.create_node("reward_field")
        self.create_node("danger_field")
        self.create_node("basin")
        self.create_node("prediction")
        self.create_node("decision")

        self.connect("vision", "reward_field")
        self.connect("vision", "danger_field")
        self.connect("reward_field", "basin")
        self.connect("danger_field", "basin")
        self.connect("basin", "prediction")
        self.connect("prediction", "decision")

    # --------------------------------------------------------

    def create_node(self, name):

        if name not in self.nodes:

            self.nodes[name] = BrainNode(name)

            self.edges[name] = []

            self.activation[name] = 0.0

    # --------------------------------------------------------

    def connect(self, a, b):

        self.create_node(a)
        self.create_node(b)

        if b not in self.edges[a]:
            self.edges[a].append(b)

    # --------------------------------------------------------

    def remove_node(self, name):

        if name in self.nodes:

            del self.nodes[name]

        if name in self.edges:
            del self.edges[name]

        for edges in self.edges.values():

            if name in edges:
                edges.remove(name)

        if name in self.activation:
            del self.activation[name]

    # --------------------------------------------------------

    def activate(self, name, value):

        if name in self.activation:

            self.activation[name] = value

            self.nodes[name].utility *= 0.99
            self.nodes[name].utility += abs(value) * 0.01

    # --------------------------------------------------------

    def adapt(self, prediction_error, danger_signal):

        """
        Polymorphic structural adaptation.

        Large prediction errors cause the system to grow
        specialized modules.

        Repeatedly useless modules can later be pruned.
        """

        if prediction_error > 2.5:

            if danger_signal > 0.5:

                self.create_node("threat_memory")
                self.connect(
                    "danger_field",
                    "threat_memory"
                )
                self.connect(
                    "threat_memory",
                    "prediction"
                )

            else:

                self.create_node("resource_memory")
                self.connect(
                    "reward_field",
                    "resource_memory"
                )
                self.connect(
                    "resource_memory",
                    "prediction"
                )

        if prediction_error > 4.0:

            self.create_node("counterfactual")

            self.connect(
                "prediction",
                "counterfactual"
            )

            self.connect(
                "counterfactual",
                "decision"
            )

        # Age all modules
        for node in self.nodes.values():
            node.age += 1

        # Increase utility of active pathways
        for node in self.nodes.values():

            act = abs(self.activation.get(node.name, 0))

            node.utility += act * 0.02

            node.utility *= 0.999

    # --------------------------------------------------------

    def prune(self, threshold):

        protected = {
            "vision",
            "reward_field",
            "danger_field",
            "basin",
            "prediction",
            "decision",
        }

        remove = []

        for name, node in self.nodes.items():

            if name not in protected:

                if node.utility < threshold and node.age > 100:

                    remove.append(name)

        for name in remove:
            self.remove_node(name)

    # --------------------------------------------------------

    def description(self):

        lines = []

        lines.append(
            f"Brain nodes: {len(self.nodes)}"
        )

        lines.append(
            f"Brain connections: "
            f"{sum(len(x) for x in self.edges.values())}"
        )

        lines.append(
            "Topology:"
        )

        for name, targets in self.edges.items():

            if targets:

                lines.append(
                    f"  {name} -> {', '.join(targets)}"
                )

        return "\n".join(lines)


# ============================================================
# WORLD
# ============================================================

class World:

    def __init__(self, size=15):

        self.size = size

        self.reset()

    # --------------------------------------------------------

    def reset(self):

        self.agent = (
            self.size // 2,
            self.size // 2
        )

        self.enemy = self.random_position(
            minimum_distance=5
        )

        self.shelter = (
            1,
            1
        )

        self.food = set()

        for _ in range(10):

            self.food.add(
                self.random_position()
            )

        self.energy = 50.0

        self.score = 0.0

        self.time = 0

        self.dead = False

        self.collected = 0

    # --------------------------------------------------------

    def random_position(self, minimum_distance=0):

        while True:

            p = (
                random.randrange(self.size),
                random.randrange(self.size)
            )

            if distance(p, self.agent) >= minimum_distance:

                return p

    # --------------------------------------------------------

    def inside(self, p):

        x, y = p

        return (
            0 <= x < self.size and
            0 <= y < self.size
        )

    # --------------------------------------------------------

    def move_enemy(self):

        ex, ey = self.enemy

        ax, ay = self.agent

        dx = 0
        dy = 0

        if ax > ex:
            dx = 1
        elif ax < ex:
            dx = -1

        if ay > ey:
            dy = 1
        elif ay < ey:
            dy = -1

        # Sometimes predator wanders
        if random.random() < 0.15:

            dx, dy = random.choice([
                (0, 0),
                (1, 0),
                (-1, 0),
                (0, 1),
                (0, -1)
            ])

        new_pos = (
            ex + dx,
            ey + dy
        )

        if self.inside(new_pos):
            self.enemy = new_pos

    # --------------------------------------------------------

    def step(self, action):

        reward = -0.03

        old_position = self.agent

        dx, dy = DIRECTION[action]

        new_position = (
            old_position[0] + dx,
            old_position[1] + dy
        )

        if self.inside(new_position):

            self.agent = new_position

        else:

            reward -= 1.0

        # ----------------------------------------------------
        # Eat
        # ----------------------------------------------------

        if action == "EAT":

            if self.agent in self.food:

                self.food.remove(self.agent)

                reward += 5.0

                self.energy += 20

                self.collected += 1

            else:

                reward -= 0.2

        # ----------------------------------------------------
        # Shelter
        # ----------------------------------------------------

        if self.agent == self.shelter:

            reward += 0.15

            self.energy += 0.5

        # ----------------------------------------------------
        # Predator
        # ----------------------------------------------------

        self.move_enemy()

        if distance(self.agent, self.enemy) <= 1.0:

            reward -= 8.0

            self.energy -= 15

        # ----------------------------------------------------
        # Energy
        # ----------------------------------------------------

        self.energy -= 0.25

        if self.energy <= 0:

            reward -= 10

            self.dead = True

        self.score += reward

        self.time += 1

        if self.time >= 250:

            self.dead = True

        return reward


# ============================================================
# HEAT MAP SYSTEM
# ============================================================

class HeatMaps:

    def __init__(self, world, genome):

        self.world = world

        self.genome = genome

        self.reward = self.empty_map()

        self.danger = self.empty_map()

        self.potential = self.empty_map()

        self.future = self.empty_map()

    # --------------------------------------------------------

    def empty_map(self):

        return [
            [0.0 for _ in range(self.world.size)]
            for _ in range(self.world.size)
        ]

    # --------------------------------------------------------

    def build(self):

        self.build_reward()

        self.build_danger()

        self.combine()

    # --------------------------------------------------------

    def build_reward(self):

        w = self.world

        for y in range(w.size):

            for x in range(w.size):

                p = (x, y)

                value = 0.0

                # Food attraction
                for food in w.food:

                    d = distance(p, food)

                    value += (
                        self.genome.food_drive
                        * gaussian(d, 2.5)
                        * 5.0
                    )

                # Shelter attraction
                d = distance(p, w.shelter)

                shelter_strength = 0.4

                if w.energy < 20:

                    shelter_strength = (
                        self.genome.shelter_drive
                        * 2.0
                    )

                value += (
                    shelter_strength
                    * gaussian(d, 5.0)
                )

                # Curiosity rewards distant/unknown locations
                center = w.size / 2

                edge_distance = min(
                    x,
                    y,
                    w.size - 1 - x,
                    w.size - 1 - y
                )

                exploration = (
                    (1.0 - edge_distance / center)
                    * self.genome.curiosity
                )

                value += exploration

                self.reward[y][x] = value

    # --------------------------------------------------------

    def build_danger(self):

        w = self.world

        for y in range(w.size):

            for x in range(w.size):

                p = (x, y)

                d = distance(p, w.enemy)

                danger = (
                    self.genome.danger_sensitivity
                    * gaussian(d, 2.7)
                    * 8.0
                )

                # Predict predator movement toward the agent.
                ex, ey = w.enemy
                ax, ay = w.agent

                px = ex + (
                    1 if ax > ex else
                    -1 if ax < ex else
                    0
                )

                py = ey + (
                    1 if ay > ey else
                    -1 if ay < ey else
                    0
                )

                predicted_enemy = (px, py)

                predicted_distance = distance(
                    p,
                    predicted_enemy
                )

                danger += (
                    self.genome.danger_sensitivity
                    * gaussian(
                        predicted_distance,
                        3.0
                    )
                    * 5.0
                )

                # Boundary uncertainty
                edge = min(
                    x,
                    y,
                    w.size - 1 - x,
                    w.size - 1 - y
                )

                if edge == 0:

                    danger += (
                        self.genome.uncertainty_aversion
                    )

                self.danger[y][x] = danger

    # --------------------------------------------------------

    def combine(self):

        for y in range(self.world.size):

            for x in range(self.world.size):

                self.potential[y][x] = (
                    self.reward[y][x]
                    -
                    self.genome.danger_weight
                    * self.danger[y][x]
                )


# ============================================================
# BASIN MATHEMATICS
# ============================================================

class BasinSolver:

    def __init__(self, maps):

        self.maps = maps

    # --------------------------------------------------------

    def neighbors(self, p):

        x, y = p

        result = []

        for action in [
            "UP",
            "DOWN",
            "LEFT",
            "RIGHT"
        ]:

            dx, dy = DIRECTION[action]

            q = (
                x + dx,
                y + dy
            )

            if (
                0 <= q[0] < self.maps.world.size
                and
                0 <= q[1] < self.maps.world.size
            ):

                result.append((action, q))

        return result

    # --------------------------------------------------------

    def gradient(self, p):

        x, y = p

        size = self.maps.world.size

        left = self.maps.potential[y][
            max(0, x - 1)
        ]

        right = self.maps.potential[y][
            min(size - 1, x + 1)
        ]

        up = self.maps.potential[
            max(0, y - 1)
        ][x]

        down = self.maps.potential[
            min(size - 1, y + 1)
        ][x]

        gx = (right - left) / 2.0

        gy = (down - up) / 2.0

        return gx, gy

    # --------------------------------------------------------

    def trace_basin(self, start, max_steps=30):

        current = start

        visited = set()

        for _ in range(max_steps):

            if current in visited:

                break

            visited.add(current)

            neighbors = self.neighbors(current)

            if not neighbors:

                break

            best = max(
                neighbors,
                key=lambda item:
                self.maps.potential[
                    item[1][1]
                ][
                    item[1][0]
                ]
            )

            current_value = self.maps.potential[
                current[1]
            ][
                current[0]
            ]

            best_value = self.maps.potential[
                best[1][1]
            ][
                best[1][0]
            ]

            if best_value <= current_value:

                break

            current = best[1]

        return current

    # --------------------------------------------------------

    def first_basin_action(self, start):

        target = self.trace_basin(start)

        if target == start:

            return "WAIT"

        x, y = start

        tx, ty = target

        if tx > x:
            return "RIGHT"

        if tx < x:
            return "LEFT"

        if ty > y:
            return "DOWN"

        if ty < y:
            return "UP"

        return "WAIT"


# ============================================================
# AGENT
# ============================================================

class Agent:

    def __init__(self, genome=None):

        self.genome = genome or Genome()

        self.brain = PolymorphicBrain()

        # Learned action values
        self.q = {
            action: 0.0
            for action in ACTIONS
        }

        # Lifetime plasticity
        self.plastic_food = 0.0
        self.plastic_danger = 0.0

        self.last_action = None

        self.last_prediction = 0.0

        self.total_error = 0.0

        self.decisions = 0

    # --------------------------------------------------------

    def effective_food_drive(self):

        return max(
            0.01,
            self.genome.food_drive
            + self.plastic_food
        )

    # --------------------------------------------------------

    def effective_danger_sensitivity(self):

        return max(
            0.01,
            self.genome.danger_sensitivity
            + self.plastic_danger
        )

    # --------------------------------------------------------

    def make_maps(self, world):

        genome = copy.deepcopy(self.genome)

        genome.food_drive = self.effective_food_drive()

        genome.danger_sensitivity = (
            self.effective_danger_sensitivity()
        )

        maps = HeatMaps(
            world,
            genome
        )

        maps.build()

        return maps

    # --------------------------------------------------------

    def predict_future_value(
        self,
        world,
        position,
        action
    ):

        """
        Short-horizon prediction.

        The agent estimates the potential after taking
        an action, then looks ahead through the local
        potential landscape.
        """

        dx, dy = DIRECTION[action]

        next_position = (
            position[0] + dx,
            position[1] + dy
        )

        if not world.inside(next_position):

            return -10.0

        maps = self.make_maps(world)

        x, y = next_position

        immediate = maps.potential[y][x]

        best_future = immediate

        current = next_position

        for _ in range(
            self.genome.prediction_horizon
        ):

            candidates = []

            cx, cy = current

            for direction in [
                "UP",
                "DOWN",
                "LEFT",
                "RIGHT"
            ]:

                ddx, ddy = DIRECTION[direction]

                q = (
                    cx + ddx,
                    cy + ddy
                )

                if world.inside(q):

                    candidates.append(
                        maps.potential[q[1]][q[0]]
                    )

            if not candidates:
                break

            best_future = max(
                best_future,
                max(candidates)
            )

            # Follow best predicted basin
            best_index = max(
                range(len(candidates)),
                key=lambda i: candidates[i]
            )

            dirs = [
                "UP",
                "DOWN",
                "LEFT",
                "RIGHT"
            ]

            ddx, ddy = DIRECTION[
                dirs[best_index]
            ]

            proposed = (
                current[0] + ddx,
                current[1] + ddy
            )

            if world.inside(proposed):

                current = proposed

        return (
            immediate
            +
            self.genome.discount
            * best_future
        )

    # --------------------------------------------------------

    def choose_action(self, world):

        maps = self.make_maps(world)

        solver = BasinSolver(maps)

        position = world.agent

        gx, gy = solver.gradient(position)

        basin_action = solver.first_basin_action(
            position
        )

        self.brain.activate(
            "vision",
            1.0
        )

        self.brain.activate(
            "reward_field",
            maps.reward[position[1]][position[0]]
        )

        self.brain.activate(
            "danger_field",
            maps.danger[position[1]][position[0]]
        )

        self.brain.activate(
            "basin",
            math.sqrt(gx * gx + gy * gy)
        )

        action_scores = []

        for action in ACTIONS:

            future = self.predict_future_value(
                world,
                position,
                action
            )

            learned = self.q[action]

            score = (
                future
                +
                learned
            )

            # Fast basin signal
            if action == basin_action:

                score += 1.5

            # Eating is only useful where food exists
            if action == "EAT":

                if position in world.food:

                    score += 5.0

                else:

                    score -= 1.0

            action_scores.append(score)

        probabilities = softmax(
            action_scores,
            self.genome.temperature
        )

        action = random.choices(
            ACTIONS,
            weights=probabilities
        )[0]

        # Slightly reduce randomness once the basin
        # direction is extremely clear.
        if random.random() < 0.15:

            action = basin_action

        self.last_action = action

        self.last_prediction = max(
            action_scores
        )

        self.decisions += 1

        return action

    # --------------------------------------------------------

    def learn(
        self,
        reward,
        next_world,
        old_world
    ):

        if self.last_action is None:
            return

        next_maps = self.make_maps(
            next_world
        )

        x, y = next_world.agent

        future_value = max(
            next_maps.potential[y][x],
            max(self.q.values())
        )

        prediction = self.q[
            self.last_action
        ]

        target = (
            reward
            +
            self.genome.discount
            * future_value
        )

        error = target - prediction

        self.q[
            self.last_action
        ] += (
            self.genome.learning_rate
            * error
        )

        self.total_error += abs(error)

        # Plasticity changes the underlying
        # behavioral parameters during life.

        if reward > 1:

            self.plastic_food += (
                0.002 * reward
            )

        if reward < -1:

            self.plastic_danger += (
                0.003 * abs(reward)
            )

        self.plastic_food = clamp(
            self.plastic_food,
            -1.0,
            1.5
        )

        self.plastic_danger = clamp(
            self.plastic_danger,
            -1.0,
            2.0
        )

        maps = self.make_maps(
            old_world
        )

        danger_signal = maps.danger[
            old_world.agent[1]
        ][
            old_world.agent[0]
        ]

        self.brain.adapt(
            abs(error),
            danger_signal
        )

        self.brain.prune(
            self.genome.prune_threshold
        )

    # --------------------------------------------------------

    def explain(self, world):

        maps = self.make_maps(world)

        position = world.agent

        solver = BasinSolver(maps)

        gx, gy = solver.gradient(position)

        basin = solver.trace_basin(
            position
        )

        return f"""
CURRENT AGENT STATE
-------------------
Position: {position}
Energy: {world.energy:.2f}
Score: {world.score:.2f}
Food collected: {world.collected}

FIELD STATE
-----------
Reward R(x,y):
{maps.reward[position[1]][position[0]]:.3f}

Danger D(x,y):
{maps.danger[position[1]][position[0]]:.3f}

Combined potential:
P = R - lambda*D

P(position):
{maps.potential[position[1]][position[0]]:.3f}

Gradient:
gx = {gx:.3f}
gy = {gy:.3f}

Current basin attractor:
{basin}

Brain nodes:
{len(self.brain.nodes)}

Brain connections:
{sum(len(x) for x in self.brain.edges.values())}
"""

    # --------------------------------------------------------

    def answer(self, question, world):

        q = question.lower()

        if "brain" in q:

            return self.brain.description()

        if "genome" in q:

            return (
                "Genome:\n"
                +
                "\n".join(
                    f"{k}: {v}"
                    for k, v in vars(
                        self.genome
                    ).items()
                )
            )

        if "reward" in q:

            maps = self.make_maps(world)

            return (
                "The reward heat map represents "
                "estimated desirability at every "
                "location. Food creates positive "
                "Gaussian fields and shelter becomes "
                "more valuable when energy is low.\n\n"
                f"Current reward value: "
                f"{maps.reward[world.agent[1]][world.agent[0]]:.3f}"
            )

        if "danger" in q or "threat" in q:

            maps = self.make_maps(world)

            return (
                "The danger heat map estimates future "
                "risk at every location, including the "
                "predicted predator position.\n\n"
                f"Current danger value: "
                f"{maps.danger[world.agent[1]][world.agent[0]]:.3f}"
            )

        if "basin" in q:

            maps = self.make_maps(world)

            solver = BasinSolver(maps)

            target = solver.trace_basin(
                world.agent
            )

            return (
                "A basin is a region of the potential "
                "landscape whose local gradient leads "
                "toward the same attractor.\n\n"
                f"Current attractor: {target}\n"
                f"First basin action: "
                f"{solver.first_basin_action(world.agent)}"
            )

        if "flee" in q or "enemy" in q:

            return (
                "I increase the danger contribution "
                "D(x,y), subtract it from reward through "
                "P = R - lambda*D, and therefore move "
                "toward regions whose combined potential "
                "is safer."
            )

        if "food" in q:

            return (
                "Food creates positive reward basins. "
                "The food-drive genome controls their "
                "strength. The agent can also learn a "
                "plastic increase in food attraction."
            )

        if "learn" in q or "training" in q:

            return (
                f"Decisions: {self.decisions}\n"
                f"Accumulated prediction error: "
                f"{self.total_error:.3f}\n"
                f"Plastic food adjustment: "
                f"{self.plastic_food:.3f}\n"
                f"Plastic danger adjustment: "
                f"{self.plastic_danger:.3f}"
            )

        if "what do you know" in q or "state" in q:

            return self.explain(world)

        return (
            "I understand questions about my "
            "brain, genome, reward map, danger map, "
            "basins, food, enemies, learning, training, "
            "and current state."
        )


# ============================================================
# EVOLUTION
# ============================================================

class Evolution:

    def __init__(
        self,
        population_size=20
    ):

        self.population_size = population_size

        self.population = [
            Agent()
            for _ in range(population_size)
        ]

        self.generation = 0

    # --------------------------------------------------------

    def episode(self, agent):

        world = World()

        while not world.dead:

            old_world = copy.deepcopy(world)

            action = agent.choose_action(
                world
            )

            reward = world.step(
                action
            )

            agent.learn(
                reward,
                world,
                old_world
            )

        fitness = (
            world.score
            +
            world.collected * 2.0
            +
            world.time * 0.03
        )

        return fitness

    # --------------------------------------------------------

    def train_generation(self):

        results = []

        for agent in self.population:

            # Reset lifetime plasticity between generations
            agent.plastic_food = 0
            agent.plastic_danger = 0

            fitness = self.episode(
                agent
            )

            results.append(
                (fitness, agent)
            )

        results.sort(
            key=lambda x: x[0],
            reverse=True
        )

        survivors = results[
            :max(
                2,
                self.population_size // 4
            )
        ]

        new_population = [
            copy.deepcopy(agent)
            for _, agent in survivors
        ]

        while len(new_population) < self.population_size:

            _, parent_a = random.choice(
                survivors
            )

            _, parent_b = random.choice(
                survivors
            )

            genome = crossover(
                parent_a.genome,
                parent_b.genome
            )

            child = Agent(genome)

            new_population.append(
                child
            )

        self.population = new_population

        self.generation += 1

        return results

    # --------------------------------------------------------

    def train(self, generations=20):

        print(
            f"\nTraining {generations} generations..."
        )

        for _ in range(generations):

            results = self.train_generation()

            fitness_values = [
                x[0]
                for x in results
            ]

            mean_fitness = (
                sum(fitness_values)
                /
                len(fitness_values)
            )

            best_fitness = max(
                fitness_values
            )

            best_agent = results[0][1]

            print(
                f"Generation "
                f"{self.generation:03d} | "
                f"mean={mean_fitness:7.2f} | "
                f"best={best_fitness:7.2f} | "
                f"brain={len(best_agent.brain.nodes)} "
                f"nodes"
            )

        print("\nTraining complete.")

    # --------------------------------------------------------

    def best_agent(self):

        return self.population[0]


# ============================================================
# ASCII VISUALIZATION
# ============================================================

def print_world(world):

    print("\nWORLD")

    for y in range(world.size):

        row = ""

        for x in range(world.size):

            p = (x, y)

            if p == world.agent:
                char = "A"

            elif p == world.enemy:
                char = "X"

            elif p == world.shelter:
                char = "S"

            elif p in world.food:
                char = "*"

            else:
                char = "."

            row += char + " "

        print(row)


def print_heatmap(grid, world, title):

    print(f"\n{title}")

    maximum = max(
        max(row)
        for row in grid
    )

    minimum = min(
        min(row)
        for row in grid
    )

    chars = " .:-=+*#%@"

    for y in range(world.size):

        row = ""

        for x in range(world.size):

            value = grid[y][x]

            if maximum == minimum:

                index = 0

            else:

                normalized = (
                    (value - minimum)
                    /
                    (maximum - minimum)
                )

                index = int(
                    normalized
                    * (len(chars) - 1)
                )

            row += chars[index]

        print(row)


# ============================================================
# TEST TRAINED AGENT
# ============================================================

def test_agent(agent):

    world = World()

    print(
        "\nRunning trained agent..."
    )

    for step in range(100):

        if world.dead:
            break

        action = agent.choose_action(
            world
        )

        reward = world.step(
            action
        )

        print(
            f"step={step:03d} "
            f"pos={world.agent} "
            f"action={action:5s} "
            f"reward={reward:6.2f} "
            f"energy={world.energy:6.2f}"
        )

        time.sleep(0.01)

    print(
        "\nFinal score:",
        round(world.score, 2)
    )

    return world


# ============================================================
# INTERACTIVE SHELL
# ============================================================

def interactive(agent):

    world = World()

    print(
        """
============================================================
PGBA INTERACTIVE AGENT
============================================================

Commands:

    train N
    test
    world
    reward
    danger
    potential
    brain
    genome
    state
    ask <question>
    step
    help
    quit

Examples:

    ask why did you flee?
    ask describe your brain
    ask what is my reward map?
    ask what is a basin?
    ask what have you learned?
"""
    )

    while True:

        try:

            command = input(
                "\nPGBA> "
            ).strip()

        except EOFError:

            break

        if not command:
            continue

        lower = command.lower()

        # ----------------------------------------------------

        if lower == "quit":

            break

        # ----------------------------------------------------

        elif lower == "help":

            print(
                "Use: train 10, test, world, reward, "
                "danger, potential, brain, genome, "
                "state, ask <question>, step."
            )

        # ----------------------------------------------------

        elif lower.startswith("train"):

            parts = command.split()

            amount = 5

            if len(parts) > 1:

                try:
                    amount = int(parts[1])
                except ValueError:
                    pass

            evolution.train(amount)

            agent = evolution.best_agent()

            world = World()

        # ----------------------------------------------------

        elif lower == "test":

            world = test_agent(
                agent
            )

        # ----------------------------------------------------

        elif lower == "step":

            if world.dead:

                world = World()

            old_world = copy.deepcopy(
                world
            )

            action = agent.choose_action(
                world
            )

            reward = world.step(
                action
            )

            agent.learn(
                reward,
                world,
                old_world
            )

            print(
                f"Action={action} "
                f"Reward={reward:.2f} "
                f"Position={world.agent} "
                f"Energy={world.energy:.2f}"
            )

        # ----------------------------------------------------

        elif lower == "world":

            print_world(world)

        # ----------------------------------------------------

        elif lower == "reward":

            maps = agent.make_maps(
                world
            )

            print_heatmap(
                maps.reward,
                world,
                "REWARD HEAT MAP R(x,y)"
            )

        # ----------------------------------------------------

        elif lower == "danger":

            maps = agent.make_maps(
                world
            )

            print_heatmap(
                maps.danger,
                world,
                "DANGER HEAT MAP D(x,y)"
            )

        # ----------------------------------------------------

        elif lower == "potential":

            maps = agent.make_maps(
                world
            )

            print_heatmap(
                maps.potential,
                world,
                "COMBINED POTENTIAL P = R - lambda*D"
            )

        # ----------------------------------------------------

        elif lower == "brain":

            print(
                agent.brain.description()
            )

        # ----------------------------------------------------

        elif lower == "genome":

            for k, v in vars(
                agent.genome
            ).items():

                print(
                    f"{k:25s} {v}"
                )

        # ----------------------------------------------------

        elif lower == "state":

            print(
                agent.explain(world)
            )

        # ----------------------------------------------------

        elif lower.startswith("ask "):

            question = command[4:]

            print(
                "\n" +
                agent.answer(
                    question,
                    world
                )
            )

        # ----------------------------------------------------

        else:

            print(
                agent.answer(
                    command,
                    world
                )
            )


# ============================================================
# MAIN
# ============================================================

if __name__ == "__main__":

    random.seed()

    print(
        """
============================================================
POLYMORPHIC GENETIC BASIN AI
============================================================

Starting evolutionary training...
"""
    )

    evolution = Evolution(
        population_size=20
    )

    # Initial training
    evolution.train(
        generations=20
    )

    agent = evolution.best_agent()

    print(
        "\nBest evolved genome:"
    )

    for key, value in vars(
        agent.genome
    ).items():

        print(
            f"  {key:25s}: {value}"
        )

    print(
        "\nEvolved brain:"
    )

    print(
        agent.brain.description()
    )

    # Start interaction
    interactive(agent)

Share this Paste