Files
emotes/tools/capture/measure.py
T
selimaj-devandclaude df9a2587e6 Add a tool that measures emotes from a store's 3D preview
tools/capture drives a store page's skin viewer in headless Firefox, stepping its clock one tick at
a time and taking every tick from ten fixed cameras. It then fits our own rig to the frames by
rendering the model (a Python port of LimbBend and PoseApplier) and matching outlines and colours,
and writes the result as Emotecraft JSON. It measures pixels only and never reads the page's
animation data. The README covers the steps, the checks and the limits.

Co-Authored-By: Claude Opus 5.5 <[email protected]>
2026-09-29 16:55:08 +02:00

342 lines
16 KiB
Python

"""Fits our rig to captured frames, tick by tick, by rendering it and matching outlines and colours.
.venv/bin/python measure.py <name> [--ticks 0:13] [--workers 7]
Writes frames/<name>/rig.json: the viewer calibration and one rig vector per tick (see model.RIG).
The first tick has nothing to start from, so it's searched in stages: the torso and head, then each
limb from several starts, then everything together, then its back-to-front twins (see flip.py). It
also finds the viewer's scale. Later ticks start from the tick before.
For a whole loop, that first pass is only a starting point: warm starts carry any drift from tick to
tick. Each later round smooths every channel over the loop (a few Fourier harmonics, which have to
close), and refits every tick held near that curve instead of near its neighbour.
"""
import argparse
import json
import multiprocessing
import time
from pathlib import Path
import cma
import numpy as np
from PIL import Image
from flip import twins
from model import INDEX, RIG, Model, rest_rig, rig_to_pose
from render import Capture, score
# How far CMA-ES first looks along each parameter: pixels for offsets, radians for angles.
STEP = np.array([2.0 if name.endswith(("_x", "_y", "_z")) else 0.35 for name in RIG])
CALIB = ["scale", "view_x", "view_y", "view_z"]
CALIB_STEP = np.array([0.05, 2, 2, 2])
# Only the scale is fitted. Moving the viewer, the whole player and the hips all slide the whole model,
# so no frame can tell them apart: the viewer's offset stays fixed (it centres the player on its
# origin), and fit.py stands the finished emote on the ground instead.
CALIB_FREE = [0]
# Channels a silhouette barely sees (a limb's twist, which way a straight limb would bend) drift
# unless something holds them near zero.
QUIET = np.array([1.0 if name.endswith(("_yaw", "_axis")) and not name.startswith(("root", "body")) else 0.0
for name in RIG])
QUIET_WEIGHT = 0.02
SMOOTH_WEIGHT = 0.05
# The search scores colour lightly in a few views: outlines give it a smooth slope to follow, and
# colour changes abruptly as parts slide across texture edges, which strands it. Choosing between a
# fit and its back-to-front twins (whose outlines match) is colour's job, so that comparison weighs
# colour heavily in every view.
COLOR_VIEWS = {"y000", "y090", "y180", "y270"}
JUDGE_COLOR_WEIGHT = 2.0
# Anatomy, as soft limits: in a tight pose like a crouch, many limb twists cast nearly the same outline,
# and these pick the natural one. Nothing is penalised inside the ranges.
FOLD = {"right_leg": np.pi, "left_leg": np.pi, "right_arm": 0.0, "left_arm": 0.0} # knees back, elbows forward
FOLD_RANGE = np.radians(70)
ROLL_RANGE = np.radians(60)
PRIOR_WEIGHT = 0.5
HIP_CENTRE_WEIGHT = 0.002
# Leaning the torso further forward while arching it further back looks nearly the same from every
# camera, so a small cost on the torso's bend picks the straightest torso that fits.
BODY_BEND_WEIGHT = 0.008
# Where each bend direction is kept (see canonical): half a turn either side of this. Knees fold back
# and elbows forward; a torso may hunch (0) or arch (pi), so both sit well inside its range.
AXIS_CENTRE = FOLD | {"body": np.pi / 2}
def wrap(a):
return (a + np.pi) % (2 * np.pi) - np.pi
ANGLES = np.array([not name.endswith(("_x", "_y", "_z")) for name in RIG])
def change(rig, previous):
"""How far each channel moved, the short way round for angles: canonical() keeps them within
-pi..pi, so a knee folding back straddles the edge and would look like it swung a whole turn."""
d = rig - previous
return np.where(ANGLES, wrap(d), d)
def anatomy(rig):
"""How far outside natural joint ranges the rig is (0 inside them)."""
total = 0.0
for limb, fold in FOLD.items():
bend = rig[INDEX[f"{limb}_bend"]]
# A negative bend folds the other way.
axis = rig[INDEX[f"{limb}_axis"]] + (np.pi if bend < 0 else 0)
# Which way a limb folds only matters as much as it bends.
total += np.sin(min(abs(bend), np.pi) / 2) * max(0.0, abs(wrap(axis - fold)) - FOLD_RANGE) ** 2
total += max(0.0, abs(wrap(rig[INDEX[f"{limb}_roll"]])) - ROLL_RANGE) ** 2
return PRIOR_WEIGHT * total + HIP_CENTRE_WEIGHT * rig[INDEX["hip_x"]] ** 2
# The hips carry everything, so the whole player's offset would only duplicate them; it stays at zero.
# Turning the whole player duplicates turning every part, so it's only fitted with --root, for flips
# and spins, where it keeps the keyframes simple.
FITTED = [i for i, n in enumerate(RIG) if not n.startswith("root")]
ROOT_TURN = [INDEX[n] for n in ("root_pitch", "root_yaw", "root_roll")]
LIMBS = ("right_leg", "left_leg", "right_arm", "left_arm")
ALL_PARTS = ("head", "body", "right_arm", "left_arm", "right_leg", "left_leg")
# Per-process state, set by init_worker, so the pool only receives parameter vectors.
_state = {}
def init_worker(directory, skin_path, tick):
skin = np.asarray(Image.open(skin_path).convert("RGBA"))
cap = Capture(directory)
model = Model(skin=skin)
# Where each part's points sit in model.colors, to score some of the parts on their own.
bounds = np.cumsum([0] + [len(model.local[p]) for p in model.local])
spans = {p: (bounds[i], bounds[i + 1]) for i, p in enumerate(model.local)}
_state.update(model=model, spans=spans, cameras=cap.cameras, targets=cap.targets(tick))
def loss(job):
"""job = (full vector: rig + calib, previous rig or None, parts to score[, judging]). Scoring only
some parts (the torso stage) counts just their points outside the captured outline. Judging weighs
colour heavily in every view, to pick between twins."""
v, previous, parts, *judging = job
judging = bool(judging and judging[0])
rig, calib = v[:len(RIG)], v[len(RIG):]
m = _state["model"]
posed = m.pose(rig_to_pose(rig))
parts = parts or ALL_PARTS
pts = np.concatenate([posed[p] for p in parts])
colors = np.concatenate([m.colors[slice(*_state["spans"][p])] for p in parts])
total = score(pts, calib, _state["cameras"], _state["targets"], colors=colors,
color_views=None if judging else COLOR_VIEWS, one_way=len(parts) < len(ALL_PARTS),
color_weight=JUDGE_COLOR_WEIGHT if judging else None)
total += QUIET_WEIGHT * np.sum(QUIET * np.sin(rig) ** 2)
total += anatomy(rig)
total += BODY_BEND_WEIGHT * rig[INDEX["body_bend"]] ** 2
if previous is not None:
total += SMOOTH_WEIGHT * np.sum((change(rig, previous) / STEP) ** 2) / len(rig)
return total
def fit(pool, x0, free, sigma, evals, previous=None, parts=None, popsize=24, seed=1):
"""CMA-ES over the indices in free (of rig + calib); the rest stay at x0."""
x0 = np.asarray(x0, float)
steps = np.concatenate([STEP, CALIB_STEP])[free]
def expand(sub):
v = x0.copy()
v[free] = sub
return v
es = cma.CMAEvolutionStrategy(x0[free], sigma, {
"CMA_stds": steps, "popsize": popsize, "maxfevals": evals, "verbose": -9, "seed": seed, "tolfun": 1e-4,
})
while not es.stop():
subs = es.ask()
es.tell(subs, pool.map(loss, [(expand(s), previous, parts) for s in subs]))
return expand(es.result.xbest), es.result.fbest
def canonical(v):
"""A negative bend is the same shape as a positive one folding the other way. Bend directions are
kept within half a turn of AXIS_CENTRE, so a knee folding back doesn't sit on the edge of the range
and jump a whole turn between ticks."""
v = v.copy()
for part, centre in AXIS_CENTRE.items():
b, a = INDEX[f"{part}_bend"], INDEX[f"{part}_axis"]
if v[b] < 0:
v[b], v[a] = -v[b], v[a] + np.pi
v[a] = wrap(v[a] - centre) + centre
return v
def loop_curve(rigs, harmonics):
"""Each channel of a loop's rigs kept to its first few Fourier harmonics: the loop's own smooth
path, which has to come back to where it started."""
from fit import periodic_smooth
rigs = np.array(rigs)
curve = np.empty_like(rigs)
for i in range(rigs.shape[1]):
values = np.unwrap(rigs[:, i]) if ANGLES[i] else rigs[:, i]
curve[:, i] = periodic_smooth(values, harmonics, bool(ANGLES[i]))[:-1]
return curve
# Starting points for each limb on its own: swung back, hanging, or forward, straight or folded the
# natural way. One start lands in a different local minimum every run; the best of these doesn't.
LIMB_STARTS = [(pitch, bend) for pitch in (-1.2, 0.0, 1.2) for bend in (0.0, 1.5)]
def fit_limbs(pool, x, fitted, starts, evals):
"""Fits each limb in turn with the others held, keeping the best of the starts."""
fx = None
for limb in LIMBS:
free = [i for i in fitted if RIG[i].startswith(limb)]
best = None
for pitch, bend in starts or [(None, None)]:
x0 = x.copy()
if pitch is not None:
x0[free] = 0
x0[INDEX[f"{limb}_pitch"]] = pitch
x0[INDEX[f"{limb}_bend"]] = bend
x0[INDEX[f"{limb}_axis"]] = FOLD[limb]
result = fit(pool, x0, free, 1.0 if starts else 0.4, evals)
if best is None or result[1] < best[1]:
best = result
x, fx = best
return x, fx
def refine_scale(pool, x):
"""The viewer scale on its own, once the pose explains the whole outline."""
from scipy.optimize import minimize_scalar
def f(scale):
v = x.copy()
v[len(RIG)] = scale
return pool.map(loss, [(v, None, None)])[0]
result = minimize_scalar(f, bounds=(x[len(RIG)] * 0.85, x[len(RIG)] * 1.15), method="bounded",
options={"xatol": 1e-3})
v = x.copy()
v[len(RIG)] = result.x
return v
def settle_facing(pool, x, fx, fitted, previous, log):
"""Tries each back-to-front twin of a fit, refined, and keeps the best. They cover nearly the same
outline, so the search can't get from one to another by itself; only colour decides."""
judge = lambda v: pool.map(loss, [(v, previous, None, True)])[0]
best, best_judged = (x, fx), judge(x)
for twin in twins(x[:len(RIG)]):
candidate = fit(pool, np.concatenate([twin, x[len(RIG):]]), fitted, 0.3, 1500, previous)
judged = judge(candidate[0])
if judged < best_judged:
log(f" a turned-round twin fits better: {judged:.3f} < {best_judged:.3f} (judged on colour)")
best, best_judged = candidate, judged
return best
def first_tick(pool, calib, fitted, log):
"""Tries the player facing both ways, all the way through, since no single stage can tell them
apart reliably: the torso, then each limb from several starts, then each limb again with the
others in place, then everything together. The scale is held until the pose explains the whole
outline (shrinking always helps a partly wrong pose), then fitted on its own."""
torso = [i for i in fitted if RIG[i].startswith(("root", "hip", "body", "head"))]
x = np.concatenate([rest_rig(), calib])
# The torso and head alone, kept inside the outline and matched by colour.
x, fx = fit(pool, x, torso, 1.0, 3000, parts=("head", "body"))
log(f" torso {fx:.3f}")
x, fx = fit_limbs(pool, x, fitted, LIMB_STARTS, 500)
log(f" limbs {fx:.3f}")
x, fx = fit_limbs(pool, x, fitted, None, 1000)
x, fx = fit(pool, x, fitted, 0.3, 3000)
log(f" together {fx:.3f}")
# Facing can be wrong at any stage; settle it once the pose explains the whole outline.
x, fx = settle_facing(pool, x, fx, fitted, None, log)
x, fx = fit(pool, x, fitted, 0.3, 6000)
x = refine_scale(pool, x)
return fit(pool, x, fitted, 0.2, 3000)
def main():
parser = argparse.ArgumentParser()
parser.add_argument("name")
parser.add_argument("--ticks", default="0:13")
parser.add_argument("--calib", default="0.9375,0,-1.5,0",
help="viewer scale (a starting guess) and offset (kept) as scale,x,y,z")
parser.add_argument("--workers", type=int, default=max(1, multiprocessing.cpu_count() - 1))
parser.add_argument("--rounds", type=int, default=2,
help="for a whole loop, how many times to refit every tick held near the loop's smoothed curve")
parser.add_argument("--harmonics", type=int, default=2,
help="Fourier harmonics in that curve: few, since it only decides what the frames can't")
parser.add_argument("--root", action="store_true", help="also turn the whole player (flips and spins)")
parser.add_argument("--resume", action="store_true",
help="start from the existing rig.json and only run the rounds")
args = parser.parse_args()
directory = Path("frames") / args.name
meta = json.loads((directory / "views.json").read_text())
skin_path = Path("skins") / f"{meta.get('skin', 'Steve')}.png"
if not skin_path.exists():
from skin import fetch
skin_path.write_bytes(fetch(meta["skin"])[0])
calib = [float(c) for c in args.calib.split(",")]
fitted = sorted(FITTED + (ROOT_TURN if args.root else []))
start, end = map(int, args.ticks.split(":"))
ticks = list(range(start, end))
whole_loop = meta.get("loop_ticks") == len(ticks)
out = directory / "rig.json"
by_tick = {}
def save():
results = [{k: v for k, v in by_tick[t].items() if k != "rig_array"} for t in ticks if t in by_tick]
out.write_text(json.dumps({"rig": RIG, "calib": calib, "scale": 4, "ticks": results}, indent=2))
def record(tick, x, fx, label, t0):
rig = canonical(x[:len(RIG)])
init_worker(directory, skin_path, tick)
per_view = score(np.concatenate(list(_state["model"].pose(rig_to_pose(rig)).values())), calib,
_state["cameras"], _state["targets"], True, _state["model"].colors, COLOR_VIEWS)
print(f"tick {tick}{label}: score {fx:.3f}, worst view {max(per_view, key=per_view.get)} "
f"{max(per_view.values()):.2f}, {time.time() - t0:.0f}s", flush=True)
by_tick[tick] = {"tick": tick, "rig": rig.tolist(), "rig_array": rig, "score": fx, "views": per_view}
save()
log = lambda m: print(m, flush=True)
if args.resume:
saved = json.loads(out.read_text())
calib = saved["calib"]
for t in saved["ticks"]:
by_tick[t["tick"]] = t | {"rig_array": canonical(np.array(t["rig"]))}
else:
# First pass: each tick warm-started from the one before.
previous = None
for tick in ticks:
t0 = time.time()
with multiprocessing.Pool(args.workers, init_worker, (directory, skin_path, tick)) as pool:
if previous is None:
x, fx = first_tick(pool, calib, fitted, log)
calib = [float(c) for c in x[len(RIG):]]
log("calibration: " + ", ".join(f"{name} {c:.3f}" for name, c in zip(CALIB, calib)))
else:
x, fx = fit(pool, np.concatenate([previous, calib]), fitted, 0.3, 4000, previous)
x, fx = settle_facing(pool, x, fx, fitted, previous, log)
record(tick, x, fx, "", t0)
previous = by_tick[tick]["rig_array"]
# Later rounds, for a loop: every tick held near the loop's smoothed curve. No tick depends on the
# one fitted before it, so drift can't build up, and the curve has to close.
for r in range(args.rounds if whole_loop else 0):
curve = loop_curve([by_tick[t]["rig_array"] for t in ticks], args.harmonics)
for i, tick in enumerate(ticks):
t0 = time.time()
anchor = canonical(curve[i])
with multiprocessing.Pool(args.workers, init_worker, (directory, skin_path, tick)) as pool:
# From its own fit and from the curve: the curve can pull a tick off a ridge its own
# fit slid along.
x, fx = min((fit(pool, np.concatenate([start, calib]), fitted, 0.2, 2000, anchor)
for start in (by_tick[tick]["rig_array"], anchor)), key=lambda c: c[1])
record(tick, x, fx, f" (round {r + 1})", t0)
print("wrote", directory / "rig.json")
if __name__ == "__main__":
main()