greedy_value_model
defender · family: Value-model optimisation · persona: hand-written baseline · author: Pelennor · live
File: greedy_value_model.py
The idea: Same energy terms, no search.
What it does
Greedy value model: annealed_assignment's value model with a greedy picker instead of the annealer. It exists so the ladder shows what the optimiser is worth on its own. The annealed_assignment docstring describes the model in full.
Record
- Best finish: #4 defender (2026-09-30).
- In the top five on 2 of the 11 nights it played.
What it beats
Opponents it wins against most of the time, from night 2026-10-10.
- rush: held in 10 of 10 seeds, mean score 0.80
- parked_ring_magazine_drain: held in 10 of 10 seeds, mean score 0.79
- attacker_blast_isolated_synchronized_rel: held in 10 of 10 seeds, mean score 0.79
- flanker: held in 10 of 10 seeds, mean score 0.79
- decoy_screen: held in 10 of 10 seeds, mean score 0.77
- … and 5 more.
What beats it
- phase_locked_ring: held in 0 of 10 seeds, mean score 0.11
- synchronized_ring: held in 0 of 10 seeds, mean score 0.14
- interleaved_shell_ring: held in 0 of 10 seeds, mean score 0.14
Replays
- Best: vs rush, seed 5: defended in 89 ticks, its score 0.82. rush__greedy_value_model__s5.json
- Worst: vs phase_locked_ring, seed 8: breached in 128 ticks, its score 0.09. 2026-09-30_attacker_phase_locked_ring__greedy_value_model__s8.json
Open one with python viewer/build_viewer.py <replay>, then viewer/index.html.
Source
"""Greedy value model: annealed_assignment's value model with a greedy picker instead of the
annealer. It exists so the ladder shows what the optimiser is worth on its own.
Idea: same energy terms, no search. Launch at the best-valued tracks one by one (skipping
any track inside the blast radius of one already picked) up to the launch cap, whenever the
net value is positive; fire the gun at the best net-valued track in range. build(),
energy() and every knob are identical to annealed_assignment.py, and a test keeps them
identical.
The annealed_assignment docstring describes the model in full.
"""
import math
import random
NAME = "greedy_value_model"
ROLE = "defender"
# --- knobs (the family differs only here) ---------------------------------------
HORIZON = 40.0 # ticks: urgency is 1 at impact, 0 at this time-to-impact or later
COST_SCALE = 1.0 # multiplies the interceptor price in the energy
UNKNOWN_FLOOR = 0.25 # P(striker | unknown) never drops below this
SWEEPS = 60
MAX_LAUNCH_VARS = 12
MAX_GUN_VARS = 6
CLUSTER_SHARE = 0.7
_state = {"seed": 0, "strikers": 0, "decoys": 0, "seen": {}}
def reset(seed):
_state.update(seed=int(seed or 0), strikers=0, decoys=0, seen={})
# --- the model --------------------------------------------------------------------
def _tti(t):
p, v, r = t["pos"], t["vel"], max(t["range"], 1e-6)
closing = -(p[0] * v[0] + p[1] * v[1]) / r
return r / max(closing, 0.5)
def _p_striker(t, prior):
c = t.get("cls", "unknown")
if c == "striker":
return 1.0
if c == "decoy":
return 0.02
return prior
def _learn(tracks):
for t in tracks:
c = t.get("cls")
if c in ("striker", "decoy") and t["id"] not in _state["seen"]:
_state["seen"][t["id"]] = c
_state["strikers" if c == "striker" else "decoys"] += 1
s, d = _state["strikers"], _state["decoys"]
return max(UNKNOWN_FLOOR, (s + 1.0) / (s + d + 2.0))
def build(obs):
rules = obs["rules"]
tracks = obs.get("tracks") or []
prior = _learn(tracks)
damage = float(rules.get("striker_damage", 10))
blast = float(rules.get("blast_radius", 10.0))
pk = float(rules.get("interceptor_kill_prob", 0.85))
gun_range = float(rules.get("gun_range", 100.0))
cap = int(rules.get("interceptor_launches_per_tick", 2))
stock = int(obs.get("stock", 0))
ammo = int(obs.get("ammo", 0))
price = float(rules.get("interceptor_cost", 4.0))
chased = {i.get("target") for i in (obs.get("interceptors") or [])}
base = {}
for t in tracks:
urg = max(0.0, min(1.0, (HORIZON - _tti(t)) / HORIZON))
base[t["id"]] = _p_striker(t, prior) * damage * urg
pos = {t["id"]: t["pos"] for t in tracks}
def near(a, b):
return math.dist(pos[a], pos[b]) <= blast
value = {}
for t in tracks:
extra = sum(base[u["id"]] for u in tracks
if u["id"] != t["id"] and near(t["id"], u["id"]))
value[t["id"]] = base[t["id"]] + CLUSTER_SHARE * extra
threats = sum(1 for t in tracks if base[t["id"]] > 0)
scarcity = 1.0 + 0.5 * max(0, threats - stock) / max(1, stock)
cost = COST_SCALE * price * scarcity
launch = [t for t in tracks if t["id"] not in chased and base[t["id"]] > 0]
launch.sort(key=lambda t: -value[t["id"]])
launch = launch[:MAX_LAUNCH_VARS] if stock > 0 else []
gun = [t for t in tracks if t["range"] <= gun_range and base[t["id"]] > 0]
gun.sort(key=lambda t: t["range"])
gun = gun[:MAX_GUN_VARS] if ammo > 0 else []
names = [("x", t["id"]) for t in launch] + [("y", t["id"]) for t in gun]
k = min(cap, stock)
slack = [("s", j) for j in range(k)] if launch else []
names += slack
index = {n: i for i, n in enumerate(names)}
lin = [0.0] * len(names)
quad = {}
def add(a, b, w):
i, j = index[a], index[b]
if i == j:
lin[i] += w
else:
key = (min(i, j), max(i, j))
quad[key] = quad.get(key, 0.0) + w
big = 2.0 * (max(value.values(), default=1.0) * pk + cost) + 1.0
for t in launch:
n = ("x", t["id"])
add(n, n, -pk * value[t["id"]] + cost)
for t in gun:
n = ("y", t["id"])
pg = max(0.0, float(rules.get("gun_pk_scale", 0.4)) * (1 - t["range"] / gun_range)
+ float(rules.get("gun_pk_floor", 0.05)))
add(n, n, -pg * value[t["id"]] + float(rules.get("gun_cost", 0.1)))
# launch cap as an equality with slack: A (sum x + sum s - k)^2
if launch:
group = [("x", t["id"]) for t in launch] + slack
for a in group:
add(a, a, big * (1 - 2 * k))
for i, a in enumerate(group):
for b in group[i + 1:]:
add(a, b, 2 * big)
for i, a in enumerate(gun):
for b in gun[i + 1:]:
add(("y", a["id"]), ("y", b["id"]), big)
for i, a in enumerate(launch):
for b in launch[i + 1:]:
if near(a["id"], b["id"]):
add(("x", a["id"]), ("x", b["id"]), pk * min(value[a["id"]], value[b["id"]]))
if ("y", a["id"]) in index:
add(("x", a["id"]), ("y", a["id"]), 0.8 * pk * base[a["id"]])
return names, lin, quad
def energy(bits, lin, quad):
e = sum(w for w, b in zip(lin, bits) if b)
return e + sum(w for (i, j), w in quad.items() if bits[i] and bits[j])
def anneal(lin, quad, rng, sweeps=SWEEPS):
n = len(lin)
if n == 0:
return []
nbr = [[] for _ in range(n)]
for (i, j), w in quad.items():
nbr[i].append((j, w))
nbr[j].append((i, w))
bits = [0] * n
e = 0.0
best, best_e = bits[:], e
scale = max([abs(w) for w in lin] + [abs(w) for w in quad.values()] + [1.0])
t_hi, t_lo = scale, scale * 1e-3
for s in range(sweeps):
temp = t_hi * (t_lo / t_hi) ** (s / max(1, sweeps - 1))
for i in range(n):
field = lin[i] + sum(w for j, w in nbr[i] if bits[j])
delta = -field if bits[i] else field
if delta <= 0 or rng.random() < math.exp(-delta / temp):
bits[i] ^= 1
e += delta
if e < best_e:
best, best_e = bits[:], e
return best
def greedy(names, lin, quad, obs):
"""The whole difference from annealed_assignment: pick, don't search."""
k = min(int(obs["rules"].get("interceptor_launches_per_tick", 2)), int(obs.get("stock", 0)))
bits = [0] * len(names)
xs = sorted((i for i, n in enumerate(names) if n[0] == "x" and lin[i] < 0),
key=lambda i: lin[i])
chosen = []
for i in xs:
if len(chosen) >= k:
break
if any(quad.get((min(i, j), max(i, j)), 0) > 0 for j in chosen):
continue
chosen.append(i)
for i in chosen:
bits[i] = 1
ys = sorted((i for i, n in enumerate(names) if n[0] == "y" and lin[i] < 0),
key=lambda i: lin[i])
if ys:
bits[ys[0]] = 1
return bits
# --- the policy -------------------------------------------------------------------
def _act(obs):
names, lin, quad = build(obs)
bits = greedy(names, lin, quad, obs)
cap = int(obs["rules"].get("interceptor_launches_per_tick", 2))
launches = [n[1] for n, b in zip(names, bits) if b and n[0] == "x"][:cap]
guns = [n[1] for n, b in zip(names, bits) if b and n[0] == "y"]
tracks = {t["id"]: t for t in (obs.get("tracks") or [])}
retarget = {}
wanted = [t for t in sorted(tracks.values(), key=_tti)
if t.get("cls") != "decoy" and t["id"] not in launches]
busy = {i.get("target") for i in (obs.get("interceptors") or [])
if i.get("target") in tracks}
free = [t["id"] for t in wanted if t["id"] not in busy]
for i in obs.get("interceptors") or []:
if i.get("target") not in tracks and free:
retarget[i["id"]] = free.pop(0)
return {"intercept": launches, "retarget": retarget, "gun": guns[0] if guns else None}
def act(obs):
try:
return _act(obs)
except Exception:
return {"intercept": [], "retarget": {}, "gun": None}