diff --git a/README.md b/README.md index ab9eba7..54b960f 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,8 @@ A player emote library for Saturn Client, with limb bending. `/emotes stop` and `/emotes info`. Press F5 to see yourself. It also has a Fabric client gametest that creates a world and screenshots each animation from the front and side, into `versions//build/run/clientGameTest/screenshots/`. +- `tools/capture/`: a Python tool that measures an emote from a store page's 3D preview and writes + it as Emotecraft JSON. See its README. Supported versions: 1.21.4 to 1.21.11, the same range as Saturn Client. Each version's Fabric API version is set in `stonecutter.properties.toml`. The code uses Mojang mappings. diff --git a/tools/capture/.gitignore b/tools/capture/.gitignore new file mode 100644 index 0000000..f371141 --- /dev/null +++ b/tools/capture/.gitignore @@ -0,0 +1,7 @@ +.venv/ +__pycache__/ +# Captures, fetched skins and scratch output: large, and regenerated by the scripts. +frames/ +skins/ +*.log +*.png diff --git a/tools/capture/CheckPort.java b/tools/capture/CheckPort.java new file mode 100644 index 0000000..43427cc --- /dev/null +++ b/tools/capture/CheckPort.java @@ -0,0 +1,19 @@ +import org.saturnclient.emotes.core.bend.LimbBend; + +/** Prints LimbBend results for check_port.py to compare with the Python port. */ +public class CheckPort { + public static void main(String[] args) { + float[][] cases = { {-2, 10, 0.5f, 0, 0}, {-2, 10, 1.75f, 3.5f, 0}, {0, 12, 1.2f, 2.79f, 0}, + {0, 12, 0.7f, 3.14159f, 1}, {0, 12, -0.9f, 1.0f, 1} }; + float[][] points = { {0, 0, 0}, {1.5f, 3, -2.25f}, {-2.25f, 10.25f, 2.25f}, {2, -2.25f, 1}, {0.5f, 12.25f, -1} }; + float[] out = new float[3]; + for (float[] c : cases) { + LimbBend b = c[4] == 1 ? LimbBend.anchoredAtEnd(c[0], c[1], c[2], c[3], 0.5f, 12) + : new LimbBend(c[0], c[1], c[2], c[3], 0.5f, 12); + for (float[] p : points) { + b.bendPoint(p[0], p[1], p[2], out); + System.out.printf("%f %f %f %f%n", out[0], out[1], out[2], b.angleAt(p[1])); + } + } + } +} diff --git a/tools/capture/README.md b/tools/capture/README.md new file mode 100644 index 0000000..758abeb --- /dev/null +++ b/tools/capture/README.md @@ -0,0 +1,63 @@ +# Emote capture + +Measures an emote from a store page's 3D preview and writes it as an Emotecraft emote our loader +reads. It works from pixels only: it renders our own player model in the same pose from the same +cameras and adjusts the pose until outlines and colours match. It never reads the page's animation +data. + +## Setup + +```sh +python3 -m venv .venv +.venv/bin/pip install -r requirements.txt +.venv/bin/playwright install firefox # Playwright's own Firefox, in its cache folder +``` + +## Steps + +```sh +.venv/bin/python capture.py --skin Steve # frames///.png + views.json +.venv/bin/python measure.py --ticks 0:13 # frames//rig.json (one loop: 13 ticks) +.venv/bin/python fit.py --symmetric --title "Twerk" --out .json +``` + +- **`capture.py`** opens the page in headless Firefox with `viewer.js` injected, switches the + preview to 3D and puts a skin on it. `--skin` takes one of the dialog's presets or any Minecraft + username. `viewer.js` replaces the page's clock, so frames step one tick (50 ms) at a time however + fast the emote moves. It also finds the camera through three.js's devtools hook. Each tick is taken + from 10 fixed cameras with a transparent background. It also measures the loop length from the + frames. +- **`skin.py`** fetches that username's skin from Mojang, so our model can wear the same texture. +- **`measure.py`** fits our rig to every tick. It renders the model with `model.py` (a port of + `LimbBend` and `PoseApplier`) and `render.py`, and scores it by outline distance and colour. It + searches with CMA-ES on all cores. The first tick is searched in stages: torso and head, then each + limb from several starts, then everything, then its back-to-front twins (`flip.py`). It also finds + the viewer's scale then. For a whole loop, later rounds refit every tick held near the loop's + smoothed curve, so drift can't build up. Leaning further forward while arching further back looks + almost the same from every camera, so a small cost on the torso's bend picks the straightest torso. +- **`fit.py`** smooths each channel over the loop (Fourier harmonics) and stands the emote on the + ground. It keys each channel wherever straight-line interpolation misses by more than 1° or 0.1 px. + `--symmetric` removes left/right lopsidedness (`symmetry.py`): each tick is averaged with the mirror + of the pose half a loop on, which keeps an even sway and drops any lean to one side. + +## Checking + +- `check_port.py` compares `model.py`'s bend maths with the Java classes (after + `./gradlew :core:compileJava`). +- `synth_test.py` renders our own twerk test animation as a fake capture, so `measure.py` can be + scored against a known answer. +- `symmetry.py ` lists the most lopsided channels. +- `compare.py ` puts the captured frames next to our render of the fit. +- `overlay.py ` draws both outlines on top of each other. +- `sheet.py` tiles captured frames. + +To see it in game, add the JSON to the testmod's `player_animations` and `EMOTECRAFT_EMOTES`, and add +shots to `EmotesGameTest`. + +## Limits + +- A loop takes about 45 minutes on 7 cores. +- The emote comes out with a keyframe on almost every tick for every channel. +- Other rigs differ from ours (a knee made of two rigid boxes, a torso that doesn't bend), so a fit is + our closest match, not an exact copy. +- Extreme poses can still land in a wrong answer on the first tick. `synth_test.py`'s 80° lean does. diff --git a/tools/capture/capture.py b/tools/capture/capture.py new file mode 100644 index 0000000..e43ecf1 --- /dev/null +++ b/tools/capture/capture.py @@ -0,0 +1,140 @@ +"""Captures an emote from a store page's 3D viewer, one frame per tick from several fixed cameras. + + .venv/bin/python capture.py [--seconds 6] + +Writes frames///.png (RGBA, transparent background) and frames//views.json, +which records each camera exactly so measure.py can render the same views. + +The page runs on a clock that viewer.js steps by hand, so every view of a tick shows the same moment +and nothing is missed however fast the emote moves. +""" +import argparse +import base64 +import io +import json +from pathlib import Path + +import numpy as np +from PIL import Image +from playwright.sync_api import sync_playwright + +TICK_MS = 50 +# Where the cameras look (viewer world units) and from how far. The model stands about one unit per +# model pixel, with its middle about 12 units below the viewer's origin. +TARGET = [0, -12, 0] +DISTANCE = 50 +FOV = 50 +VIEWS = {f"y{yaw:03d}": (yaw, 0) for yaw in range(0, 360, 45)} +VIEWS |= {"top000": (0, 50), "top090": (90, 50)} + + +def step(page, ms, grab=False): + url = page.evaluate(f"window.__capture.step({ms}, {'true' if grab else 'false'})") + if not grab: + return None + return Image.open(io.BytesIO(base64.b64decode(url.split(",", 1)[1]))).convert("RGBA") + + +def warm_up(page, frames): + for _ in range(frames): + step(page, 16) + page.wait_for_timeout(15) + + +def set_view(page, yaw, pitch): + view = {"target": TARGET, "distance": DISTANCE, "yaw": yaw, "pitch": pitch, "fov": FOV} + page.evaluate(f"window.__capture.setView({json.dumps(view)})") + return view + + +def choose_skin(page, skin): + """Puts a skin we have the texture of on the model, so measure.py can match colours too.""" + page.evaluate("[...document.querySelectorAll('button')].find(b => b.innerText.includes(\"'s skin\")).click()") + warm_up(page, 40) + page.get_by_text("Change preview skin").last.click(force=True) + warm_up(page, 60) + preset = page.get_by_role("button", name=skin, exact=True) + if preset.count(): + preset.first.click(force=True) + else: + page.get_by_placeholder("Enter a Minecraft username").fill(skin) + page.keyboard.press("Enter") + warm_up(page, 150) + page.keyboard.press("Escape") + + +def loop_ticks(masks): + """The loop length in ticks: the shift that best maps the front view's masks onto themselves.""" + best, best_err = None, None + for period in range(8, len(masks) // 2): + err = np.mean([np.mean(masks[i] != masks[i + period]) for i in range(len(masks) - period)]) + if best_err is None or err < best_err: + best, best_err = period, err + return best, best_err + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("url") + parser.add_argument("name") + parser.add_argument("--seconds", type=float, default=6) + parser.add_argument("--skin", default="Steve", + help="a preset from the viewer's skin dialog (Steve, Alex, ...) or a Minecraft username") + args = parser.parse_args() + + out = Path("frames") / args.name + ticks = round(args.seconds * 1000 / TICK_MS) + + with sync_playwright() as p: + browser = p.firefox.launch() + page = browser.new_page(viewport={"width": 1280, "height": 900}) + page.add_init_script(path=str(Path(__file__).with_name("viewer.js"))) + page.goto(args.url, wait_until="domcontentloaded", timeout=60000) + warm_up(page, 200) + # The toggle between the video and the 3D viewer sits in the preview's top right corner. + page.mouse.click(794, 177) + page.wait_for_selector("#skin-viewer", timeout=30000) + warm_up(page, 100) + choose_skin(page, args.skin) + warm_up(page, 200) + + cameras = {} + for name, (yaw, pitch) in VIEWS.items(): + (out / name).mkdir(parents=True, exist_ok=True) + cameras[name] = set_view(page, yaw, pitch) + + info = page.evaluate("window.__capture.camera_info()") + start = page.evaluate("window.__capture.time()") + front_masks = [] + for tick in range(ticks): + for i, (name, (yaw, pitch)) in enumerate(VIEWS.items()): + set_view(page, yaw, pitch) + # The first view advances the clock by a tick; the rest redraw the same moment. + frame = step(page, TICK_MS if i == 0 else 0, grab=True) + frame.save(out / name / f"{tick:04d}.png") + if name == "y000": + front_masks.append(np.asarray(frame)[:, :, 3] > 127) + if tick % 20 == 0: + print(f"tick {tick}/{ticks}") + + period, err = loop_ticks(front_masks) + meta = { + "url": args.url, + "skin": args.skin, + "tick_ms": TICK_MS, + "ticks": ticks, + "start_ms": start, + "size": list(front_masks[0].shape[::-1]), + "aspect": info["aspect"], + "near": info["near"], + "views": cameras, + "loop_ticks": period, + "loop_error": err, + } + (out / "views.json").write_text(json.dumps(meta, indent=2)) + print(f"loop: {period} ticks ({period * TICK_MS / 1000:.2f} s), mismatch {err:.4f}") + browser.close() + + +if __name__ == "__main__": + main() diff --git a/tools/capture/check_port.py b/tools/capture/check_port.py new file mode 100644 index 0000000..b421ce6 --- /dev/null +++ b/tools/capture/check_port.py @@ -0,0 +1,27 @@ +"""Compares model.LimbBend with the Java LimbBend, point for point. + + .venv/bin/python check_port.py ../../core/build/classes/java/main # after ./gradlew :core:compileJava + +Runs CheckPort.java against the compiled core classes and fails if any result differs by 1e-3 or more. +""" +import subprocess +import sys + +import numpy as np + +from model import LimbBend + +CASES = [(-2, 10, 0.5, 0, 0), (-2, 10, 1.75, 3.5, 0), (0, 12, 1.2, 2.79, 0), (0, 12, 0.7, 3.14159, 1), (0, 12, -0.9, 1.0, 1)] +POINTS = np.array([[0, 0, 0], [1.5, 3, -2.25], [-2.25, 10.25, 2.25], [2, -2.25, 1], [0.5, 12.25, -1]]) + +classes = sys.argv[1] +java = subprocess.run(["java", "-cp", classes, "CheckPort.java"], capture_output=True, text=True, check=True).stdout +expected = np.array([[float(v) for v in line.split()] for line in java.strip().splitlines()]) +got = [] +for start, end, angle, axis, anchored in CASES: + b = LimbBend(start, end, angle, axis, anchored_at_end=bool(anchored)) + bent = b.bend_points(POINTS) + got += [[*bent[i], b.angle_at(POINTS[i, 1])] for i in range(len(POINTS))] +err = np.abs(np.array(got) - expected).max() +print(f"max difference from Java: {err:.2e}") +sys.exit(0 if err < 1e-3 else 1) diff --git a/tools/capture/compare.py b/tools/capture/compare.py new file mode 100644 index 0000000..78f48e7 --- /dev/null +++ b/tools/capture/compare.py @@ -0,0 +1,40 @@ +"""Puts captured frames next to our render of the fitted pose, with the same skin and cameras. + + .venv/bin/python compare.py [tick] [views] + +Writes frames//compare.png: for each view, theirs on the left and ours on the right. +""" +import json +import sys +from pathlib import Path + +import numpy as np +from PIL import Image + +from model import Model, rig_to_pose +from render import Camera, render_colors + +name = sys.argv[1] +tick = int(sys.argv[2]) if len(sys.argv) > 2 else 0 +views = sys.argv[3].split(",") if len(sys.argv) > 3 else ["y000", "y090", "y180", "y270", "y045"] +base = Path("frames") / name +meta = json.loads((base / "views.json").read_text()) +data = json.loads((base / "rig.json").read_text()) +rig = np.array(next(t["rig"] for t in data["ticks"] if t["tick"] == tick)) +skin = np.asarray(Image.open(Path("skins") / f"{meta.get('skin', 'Steve')}.png").convert("RGBA")) +model = Model(spacing=0.12, skin=skin) +scale = 2 +cams = {v: Camera(meta["views"][v], meta["size"], meta["aspect"], scale) for v in views} +ours = render_colors(np.concatenate(list(model.pose(rig_to_pose(rig)).values())), model.colors, data["calib"], cams) +bg = np.array([40, 40, 52], np.uint8) +tiles = [] +for v in views: + theirs = np.asarray(Image.open(base / v / f"{tick:04d}.png").convert("RGBA").reduce(scale)) + a = np.where(theirs[..., 3:] > 127, theirs[..., :3], bg) + image, mask = ours[v] + b = np.where(mask[..., None], (image * 255).astype(np.uint8), bg) + tiles.append(np.concatenate([a, b], axis=1)) +sheet = np.concatenate(tiles, axis=0) +ys, xs = np.nonzero((sheet != bg).any(axis=2)) +Image.fromarray(sheet[:, max(xs.min() - 8, 0):xs.max() + 8]).save(base / "compare.png") +print("wrote", base / "compare.png") diff --git a/tools/capture/fit.py b/tools/capture/fit.py new file mode 100644 index 0000000..6759d09 --- /dev/null +++ b/tools/capture/fit.py @@ -0,0 +1,166 @@ +"""Turns measured rigs (frames//rig.json) into an Emotecraft emote our loader reads. + + .venv/bin/python fit.py [--out path.json] [--angle-tol 1] [--offset-tol 0.1] + +Every channel is keyed independently, keeping only the ticks linear interpolation can't reproduce +within the tolerances. Values go through the inverse of EmotecraftLoader's conversions. +""" +import argparse +import json +from pathlib import Path + +import numpy as np + +from model import INDEX, Model, rest_rig, rig_to_pose + +PARTS = {"root": "body", "head": "head", "body": "torso", "right_arm": "rightArm", "left_arm": "leftArm", + "right_leg": "rightLeg", "left_leg": "leftLeg"} +# EmotecraftLoader.REST_PIVOTS: limb positions in the file are absolute pivots. +REST = {"right_arm": (-5, 2, 0), "left_arm": (5, 2, 0), "right_leg": (-1.9, 12, 0.1), "left_leg": (1.9, 12, 0.1)} +CHANNELS = ["x", "y", "z", "pitch", "yaw", "roll", "bend", "axis"] + + +def to_file(bone, channel, value): + """The inverse of EmotecraftLoader.convert, writing degrees.""" + if channel in ("x", "y", "z"): + if bone == "root": + return value / 16 if channel == "z" else -value / 16 + return value + REST.get(bone, (0, 0, 0))["xyz".index(channel)] + degrees = np.degrees(value) + if channel == "bend": + return -degrees + if bone == "root" and channel in ("pitch", "yaw"): + return -degrees + return degrees + + +def keep(values, tol): + """Indices to key so linear interpolation stays within tol of every sample (Douglas-Peucker).""" + keys = {0, len(values) - 1} + + def split(a, b): + if b - a < 2: + return + t = np.arange(a, b + 1) + line = values[a] + (values[b] - values[a]) * (t - a) / (b - a) + err = np.abs(values[a:b + 1] - line) + i = int(np.argmax(err)) + if err[i] > tol: + keys.add(a + i) + split(a, a + i) + split(a + i, b) + + split(0, len(values) - 1) + return sorted(keys) + + +def ground(rigs): + """Shifts every tick vertically so the lowest foot over the whole emote is where a standing + player's feet are. The viewer offset was a guess (measure.py keeps it fixed), so the measured + height is only right up to a constant.""" + model = Model() + lowest = lambda rig: max(model.pose(rig_to_pose(rig))[leg][:, 1].max() for leg in ("right_leg", "left_leg")) + shift = lowest(rest_rig()) - max(lowest(r) for r in rigs) # model Y points down + for r in rigs: + r[INDEX["root_y"]] += shift + return shift + + +def periodic_smooth(values, harmonics, angle): + """A loop's channel kept to its first few Fourier harmonics: removes the fit's tick-to-tick noise + and joins the loop's seam. Returns the smoothed samples plus the loop's next sample (its start + again, a whole number of turns on for an angle that spins).""" + n = len(values) + turns = 0.0 + if angle: + # How far the angle travels in one loop: nothing for a sway, whole turns for a spin. + step = (values[0] - values[-1] + np.pi) % (2 * np.pi) - np.pi + turns = values[-1] + step - values[0] + ramp = turns * np.arange(n + 1) / n + spectrum = np.fft.rfft(values - ramp[:n]) + spectrum[harmonics + 1:] = 0 + smooth = np.fft.irfft(spectrum, n) + return np.append(smooth, smooth[0]) + ramp + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("name") + parser.add_argument("--out") + parser.add_argument("--angle-tol", type=float, default=1.0, help="degrees") + parser.add_argument("--offset-tol", type=float, default=0.1, help="pixels") + parser.add_argument("--no-ground", action="store_true", help="keep the measured height") + parser.add_argument("--harmonics", type=int, default=4, + help="for a whole loop, how many Fourier harmonics each channel keeps (0 = no smoothing)") + parser.add_argument("--rig", help="read this rig file instead of frames//rig.json") + parser.add_argument("--symmetric", action="store_true", + help="for a whole loop that sways once per loop, remove its left/right lopsidedness " + "(see symmetry.py)") + parser.add_argument("--title", help="the emote's name (default: from )") + parser.add_argument("--description", help="default: where it was measured from") + parser.add_argument("--author", default="Saturn capture") + args = parser.parse_args() + + base = Path("frames") / args.name + data = json.loads(Path(args.rig).read_text() if args.rig else (base / "rig.json").read_text()) + meta = json.loads((base / "views.json").read_text()) + ticks = data["ticks"] + loop = meta.get("loop_ticks") + rigs = [np.array(t["rig"]) for t in ticks] + if args.symmetric: + if not (loop and len(rigs) == loop): + parser.error("--symmetric needs a whole loop") + from symmetry import symmetrize + rigs = list(symmetrize(rigs, sway=True)) + print("made symmetric: each tick averaged with the mirror of the pose half a loop on") + if not args.no_ground: + print(f"grounded: moved {ground(rigs):+.2f} px down") + poses = [rig_to_pose(r) for r in rigs] + first = ticks[0]["tick"] + whole_loop = bool(loop) and len(poses) == loop + if whole_loop and args.harmonics: + print(f"smoothed each channel to {args.harmonics} harmonics of the {loop}-tick loop") + + moves = {} + total = 0 + for bone, part in PARTS.items(): + for channel in CHANNELS: + raw = np.array([getattr(p[bone], channel) for p in poses], float) + angle = channel not in ("x", "y", "z") + if angle: + raw = np.unwrap(raw) + if whole_loop: + # A loop's last tick blends back into its first. + raw = periodic_smooth(raw, args.harmonics, angle) if args.harmonics else np.append(raw, raw[0]) + values = np.array([to_file(bone, channel, v) for v in raw]) + if np.all(np.abs(values - values[0]) < 1e-6) and abs(values[0]) < 1e-6 and channel != "x": + continue + tol = args.offset_tol if channel in ("x", "y", "z") else args.angle_tol + if bone == "root" and channel in ("x", "y", "z"): + tol /= 16 + for i in keep(values, tol): + moves.setdefault(i, {}).setdefault(part, {})[channel] = round(float(values[i]), 4) + total += 1 + + emote = { + "version": 3, + "name": args.title or args.name.replace("_", " ").title(), + "description": args.description or f"Measured from {meta['url']}", + "author": args.author, + "emote": { + "degrees": True, + "isLoop": bool(loop), + "returnTick": 0, + "endTick": len(poses) - 1, + "stopTick": len(poses) if whole_loop else len(poses) - 1, + "moves": [{"tick": i, "easing": "LINEAR", **parts} for i, parts in sorted(moves.items())], + }, + } + out = Path(args.out) if args.out else base / f"{args.name}.json" + out.write_text(json.dumps(emote, indent=2)) + print(f"wrote {out}: {len(emote['emote']['moves'])} ticks keyed, {total} channel keyframes " + f"(from {len(poses)} samples, starting at captured tick {first})") + + +if __name__ == "__main__": + main() diff --git a/tools/capture/flip.py b/tools/capture/flip.py new file mode 100644 index 0000000..248362b --- /dev/null +++ b/tools/capture/flip.py @@ -0,0 +1,55 @@ +"""The back-to-front twin of a rig, which covers the same space with the textures turned round. + +The head and torso boxes are centred on their pivots, so spinning either half a turn around its own +vertical axis leaves it covering the same space; only which face shows changes. The arms hang from +the torso's sides, so spinning the torso carries each arm to the other side. That twin (torso and +head facing backwards, arms swapped) is a separate local minimum the search can't walk out of, so +measure.py jumps to it directly, refines it, and keeps whichever scores lower. Only the colours tell +them apart. + +In rig terms: the torso and head get R Ry(pi); each arm takes the other's rotation times Ry(pi) (a +half turn around its own length, which moves its off-centre box to the other side of its pivot); and +bends keep their shape with their direction turned half round, since the frame they're measured in +turned. The legs hang from the hips and don't change. +""" +import numpy as np + +from model import INDEX, rotation_zyx + +SPIN = np.diag([-1.0, 1.0, -1.0]) # Ry(pi) +ARMS = {"right_arm": "left_arm", "left_arm": "right_arm"} + + +def euler_zyx(r): + """(pitch, yaw, roll) with rotation_zyx(pitch, yaw, roll) == r.""" + yaw = np.arcsin(np.clip(-r[2, 0], -1, 1)) + return np.arctan2(r[2, 1], r[2, 2]), yaw, np.arctan2(r[1, 0], r[0, 0]) + + +def _spun(rig, source): + g = lambda c: rig[INDEX[f"{source}_{c}"]] + return euler_zyx(rotation_zyx(g("pitch"), g("yaw"), g("roll")) @ SPIN) + + +def flip_head(rig): + """Just the head spun round: a cube on its pivot, so the same space facing the other way.""" + out = rig.copy() + for c, v in zip(("pitch", "yaw", "roll"), _spun(rig, "head")): + out[INDEX[f"head_{c}"]] = v + return out + + +def twins(rig): + """Every back-to-front twin worth trying: the torso with its arms, the head, and both.""" + return [flip(rig), flip_head(rig), flip_head(flip(rig))] + + +def flip(rig): + out = rig.copy() + for part, source in [("body", "body"), ("head", "head"), *ARMS.items()]: + for c, v in zip(("pitch", "yaw", "roll"), _spun(rig, source)): + out[INDEX[f"{part}_{c}"]] = v + for part, source in [("body", "body"), *ARMS.items()]: + out[INDEX[f"{part}_bend"]] = rig[INDEX[f"{source}_bend"]] + out[INDEX[f"{part}_axis"]] = rig[INDEX[f"{source}_axis"]] + np.pi + return out diff --git a/tools/capture/measure.py b/tools/capture/measure.py new file mode 100644 index 0000000..b5f0716 --- /dev/null +++ b/tools/capture/measure.py @@ -0,0 +1,341 @@ +"""Fits our rig to captured frames, tick by tick, by rendering it and matching outlines and colours. + + .venv/bin/python measure.py [--ticks 0:13] [--workers 7] + +Writes frames//rig.json: the viewer calibration and one rig vector per tick (see model.RIG). + +The first tick has nothing to start from, so it's searched in stages: the torso and head, then each +limb from several starts, then everything together, then its back-to-front twins (see flip.py). It +also finds the viewer's scale. Later ticks start from the tick before. + +For a whole loop, that first pass is only a starting point: warm starts carry any drift from tick to +tick. Each later round smooths every channel over the loop (a few Fourier harmonics, which have to +close), and refits every tick held near that curve instead of near its neighbour. +""" +import argparse +import json +import multiprocessing +import time +from pathlib import Path + +import cma +import numpy as np +from PIL import Image + +from flip import twins +from model import INDEX, RIG, Model, rest_rig, rig_to_pose +from render import Capture, score + +# How far CMA-ES first looks along each parameter: pixels for offsets, radians for angles. +STEP = np.array([2.0 if name.endswith(("_x", "_y", "_z")) else 0.35 for name in RIG]) +CALIB = ["scale", "view_x", "view_y", "view_z"] +CALIB_STEP = np.array([0.05, 2, 2, 2]) +# Only the scale is fitted. Moving the viewer, the whole player and the hips all slide the whole model, +# so no frame can tell them apart: the viewer's offset stays fixed (it centres the player on its +# origin), and fit.py stands the finished emote on the ground instead. +CALIB_FREE = [0] +# Channels a silhouette barely sees (a limb's twist, which way a straight limb would bend) drift +# unless something holds them near zero. +QUIET = np.array([1.0 if name.endswith(("_yaw", "_axis")) and not name.startswith(("root", "body")) else 0.0 + for name in RIG]) +QUIET_WEIGHT = 0.02 +SMOOTH_WEIGHT = 0.05 +# The search scores colour lightly in a few views: outlines give it a smooth slope to follow, and +# colour changes abruptly as parts slide across texture edges, which strands it. Choosing between a +# fit and its back-to-front twins (whose outlines match) is colour's job, so that comparison weighs +# colour heavily in every view. +COLOR_VIEWS = {"y000", "y090", "y180", "y270"} +JUDGE_COLOR_WEIGHT = 2.0 + +# Anatomy, as soft limits: in a tight pose like a crouch, many limb twists cast nearly the same outline, +# and these pick the natural one. Nothing is penalised inside the ranges. +FOLD = {"right_leg": np.pi, "left_leg": np.pi, "right_arm": 0.0, "left_arm": 0.0} # knees back, elbows forward +FOLD_RANGE = np.radians(70) +ROLL_RANGE = np.radians(60) +PRIOR_WEIGHT = 0.5 +HIP_CENTRE_WEIGHT = 0.002 +# Leaning the torso further forward while arching it further back looks nearly the same from every +# camera, so a small cost on the torso's bend picks the straightest torso that fits. +BODY_BEND_WEIGHT = 0.008 +# Where each bend direction is kept (see canonical): half a turn either side of this. Knees fold back +# and elbows forward; a torso may hunch (0) or arch (pi), so both sit well inside its range. +AXIS_CENTRE = FOLD | {"body": np.pi / 2} + + +def wrap(a): + return (a + np.pi) % (2 * np.pi) - np.pi + + +ANGLES = np.array([not name.endswith(("_x", "_y", "_z")) for name in RIG]) + + +def change(rig, previous): + """How far each channel moved, the short way round for angles: canonical() keeps them within + -pi..pi, so a knee folding back straddles the edge and would look like it swung a whole turn.""" + d = rig - previous + return np.where(ANGLES, wrap(d), d) + + +def anatomy(rig): + """How far outside natural joint ranges the rig is (0 inside them).""" + total = 0.0 + for limb, fold in FOLD.items(): + bend = rig[INDEX[f"{limb}_bend"]] + # A negative bend folds the other way. + axis = rig[INDEX[f"{limb}_axis"]] + (np.pi if bend < 0 else 0) + # Which way a limb folds only matters as much as it bends. + total += np.sin(min(abs(bend), np.pi) / 2) * max(0.0, abs(wrap(axis - fold)) - FOLD_RANGE) ** 2 + total += max(0.0, abs(wrap(rig[INDEX[f"{limb}_roll"]])) - ROLL_RANGE) ** 2 + return PRIOR_WEIGHT * total + HIP_CENTRE_WEIGHT * rig[INDEX["hip_x"]] ** 2 + +# The hips carry everything, so the whole player's offset would only duplicate them; it stays at zero. +# Turning the whole player duplicates turning every part, so it's only fitted with --root, for flips +# and spins, where it keeps the keyframes simple. +FITTED = [i for i, n in enumerate(RIG) if not n.startswith("root")] +ROOT_TURN = [INDEX[n] for n in ("root_pitch", "root_yaw", "root_roll")] +LIMBS = ("right_leg", "left_leg", "right_arm", "left_arm") +ALL_PARTS = ("head", "body", "right_arm", "left_arm", "right_leg", "left_leg") + +# Per-process state, set by init_worker, so the pool only receives parameter vectors. +_state = {} + + +def init_worker(directory, skin_path, tick): + skin = np.asarray(Image.open(skin_path).convert("RGBA")) + cap = Capture(directory) + model = Model(skin=skin) + # Where each part's points sit in model.colors, to score some of the parts on their own. + bounds = np.cumsum([0] + [len(model.local[p]) for p in model.local]) + spans = {p: (bounds[i], bounds[i + 1]) for i, p in enumerate(model.local)} + _state.update(model=model, spans=spans, cameras=cap.cameras, targets=cap.targets(tick)) + + +def loss(job): + """job = (full vector: rig + calib, previous rig or None, parts to score[, judging]). Scoring only + some parts (the torso stage) counts just their points outside the captured outline. Judging weighs + colour heavily in every view, to pick between twins.""" + v, previous, parts, *judging = job + judging = bool(judging and judging[0]) + rig, calib = v[:len(RIG)], v[len(RIG):] + m = _state["model"] + posed = m.pose(rig_to_pose(rig)) + parts = parts or ALL_PARTS + pts = np.concatenate([posed[p] for p in parts]) + colors = np.concatenate([m.colors[slice(*_state["spans"][p])] for p in parts]) + total = score(pts, calib, _state["cameras"], _state["targets"], colors=colors, + color_views=None if judging else COLOR_VIEWS, one_way=len(parts) < len(ALL_PARTS), + color_weight=JUDGE_COLOR_WEIGHT if judging else None) + total += QUIET_WEIGHT * np.sum(QUIET * np.sin(rig) ** 2) + total += anatomy(rig) + total += BODY_BEND_WEIGHT * rig[INDEX["body_bend"]] ** 2 + if previous is not None: + total += SMOOTH_WEIGHT * np.sum((change(rig, previous) / STEP) ** 2) / len(rig) + return total + + +def fit(pool, x0, free, sigma, evals, previous=None, parts=None, popsize=24, seed=1): + """CMA-ES over the indices in free (of rig + calib); the rest stay at x0.""" + x0 = np.asarray(x0, float) + steps = np.concatenate([STEP, CALIB_STEP])[free] + + def expand(sub): + v = x0.copy() + v[free] = sub + return v + + es = cma.CMAEvolutionStrategy(x0[free], sigma, { + "CMA_stds": steps, "popsize": popsize, "maxfevals": evals, "verbose": -9, "seed": seed, "tolfun": 1e-4, + }) + while not es.stop(): + subs = es.ask() + es.tell(subs, pool.map(loss, [(expand(s), previous, parts) for s in subs])) + return expand(es.result.xbest), es.result.fbest + + +def canonical(v): + """A negative bend is the same shape as a positive one folding the other way. Bend directions are + kept within half a turn of AXIS_CENTRE, so a knee folding back doesn't sit on the edge of the range + and jump a whole turn between ticks.""" + v = v.copy() + for part, centre in AXIS_CENTRE.items(): + b, a = INDEX[f"{part}_bend"], INDEX[f"{part}_axis"] + if v[b] < 0: + v[b], v[a] = -v[b], v[a] + np.pi + v[a] = wrap(v[a] - centre) + centre + return v + + +def loop_curve(rigs, harmonics): + """Each channel of a loop's rigs kept to its first few Fourier harmonics: the loop's own smooth + path, which has to come back to where it started.""" + from fit import periodic_smooth + rigs = np.array(rigs) + curve = np.empty_like(rigs) + for i in range(rigs.shape[1]): + values = np.unwrap(rigs[:, i]) if ANGLES[i] else rigs[:, i] + curve[:, i] = periodic_smooth(values, harmonics, bool(ANGLES[i]))[:-1] + return curve + + +# Starting points for each limb on its own: swung back, hanging, or forward, straight or folded the +# natural way. One start lands in a different local minimum every run; the best of these doesn't. +LIMB_STARTS = [(pitch, bend) for pitch in (-1.2, 0.0, 1.2) for bend in (0.0, 1.5)] + + +def fit_limbs(pool, x, fitted, starts, evals): + """Fits each limb in turn with the others held, keeping the best of the starts.""" + fx = None + for limb in LIMBS: + free = [i for i in fitted if RIG[i].startswith(limb)] + best = None + for pitch, bend in starts or [(None, None)]: + x0 = x.copy() + if pitch is not None: + x0[free] = 0 + x0[INDEX[f"{limb}_pitch"]] = pitch + x0[INDEX[f"{limb}_bend"]] = bend + x0[INDEX[f"{limb}_axis"]] = FOLD[limb] + result = fit(pool, x0, free, 1.0 if starts else 0.4, evals) + if best is None or result[1] < best[1]: + best = result + x, fx = best + return x, fx + + +def refine_scale(pool, x): + """The viewer scale on its own, once the pose explains the whole outline.""" + from scipy.optimize import minimize_scalar + + def f(scale): + v = x.copy() + v[len(RIG)] = scale + return pool.map(loss, [(v, None, None)])[0] + + result = minimize_scalar(f, bounds=(x[len(RIG)] * 0.85, x[len(RIG)] * 1.15), method="bounded", + options={"xatol": 1e-3}) + v = x.copy() + v[len(RIG)] = result.x + return v + + +def settle_facing(pool, x, fx, fitted, previous, log): + """Tries each back-to-front twin of a fit, refined, and keeps the best. They cover nearly the same + outline, so the search can't get from one to another by itself; only colour decides.""" + judge = lambda v: pool.map(loss, [(v, previous, None, True)])[0] + best, best_judged = (x, fx), judge(x) + for twin in twins(x[:len(RIG)]): + candidate = fit(pool, np.concatenate([twin, x[len(RIG):]]), fitted, 0.3, 1500, previous) + judged = judge(candidate[0]) + if judged < best_judged: + log(f" a turned-round twin fits better: {judged:.3f} < {best_judged:.3f} (judged on colour)") + best, best_judged = candidate, judged + return best + + +def first_tick(pool, calib, fitted, log): + """Tries the player facing both ways, all the way through, since no single stage can tell them + apart reliably: the torso, then each limb from several starts, then each limb again with the + others in place, then everything together. The scale is held until the pose explains the whole + outline (shrinking always helps a partly wrong pose), then fitted on its own.""" + torso = [i for i in fitted if RIG[i].startswith(("root", "hip", "body", "head"))] + x = np.concatenate([rest_rig(), calib]) + # The torso and head alone, kept inside the outline and matched by colour. + x, fx = fit(pool, x, torso, 1.0, 3000, parts=("head", "body")) + log(f" torso {fx:.3f}") + x, fx = fit_limbs(pool, x, fitted, LIMB_STARTS, 500) + log(f" limbs {fx:.3f}") + x, fx = fit_limbs(pool, x, fitted, None, 1000) + x, fx = fit(pool, x, fitted, 0.3, 3000) + log(f" together {fx:.3f}") + # Facing can be wrong at any stage; settle it once the pose explains the whole outline. + x, fx = settle_facing(pool, x, fx, fitted, None, log) + x, fx = fit(pool, x, fitted, 0.3, 6000) + x = refine_scale(pool, x) + return fit(pool, x, fitted, 0.2, 3000) + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("name") + parser.add_argument("--ticks", default="0:13") + parser.add_argument("--calib", default="0.9375,0,-1.5,0", + help="viewer scale (a starting guess) and offset (kept) as scale,x,y,z") + parser.add_argument("--workers", type=int, default=max(1, multiprocessing.cpu_count() - 1)) + parser.add_argument("--rounds", type=int, default=2, + help="for a whole loop, how many times to refit every tick held near the loop's smoothed curve") + parser.add_argument("--harmonics", type=int, default=2, + help="Fourier harmonics in that curve: few, since it only decides what the frames can't") + parser.add_argument("--root", action="store_true", help="also turn the whole player (flips and spins)") + parser.add_argument("--resume", action="store_true", + help="start from the existing rig.json and only run the rounds") + args = parser.parse_args() + + directory = Path("frames") / args.name + meta = json.loads((directory / "views.json").read_text()) + skin_path = Path("skins") / f"{meta.get('skin', 'Steve')}.png" + if not skin_path.exists(): + from skin import fetch + skin_path.write_bytes(fetch(meta["skin"])[0]) + calib = [float(c) for c in args.calib.split(",")] + fitted = sorted(FITTED + (ROOT_TURN if args.root else [])) + start, end = map(int, args.ticks.split(":")) + + ticks = list(range(start, end)) + whole_loop = meta.get("loop_ticks") == len(ticks) + out = directory / "rig.json" + by_tick = {} + + def save(): + results = [{k: v for k, v in by_tick[t].items() if k != "rig_array"} for t in ticks if t in by_tick] + out.write_text(json.dumps({"rig": RIG, "calib": calib, "scale": 4, "ticks": results}, indent=2)) + + def record(tick, x, fx, label, t0): + rig = canonical(x[:len(RIG)]) + init_worker(directory, skin_path, tick) + per_view = score(np.concatenate(list(_state["model"].pose(rig_to_pose(rig)).values())), calib, + _state["cameras"], _state["targets"], True, _state["model"].colors, COLOR_VIEWS) + print(f"tick {tick}{label}: score {fx:.3f}, worst view {max(per_view, key=per_view.get)} " + f"{max(per_view.values()):.2f}, {time.time() - t0:.0f}s", flush=True) + by_tick[tick] = {"tick": tick, "rig": rig.tolist(), "rig_array": rig, "score": fx, "views": per_view} + save() + + log = lambda m: print(m, flush=True) + if args.resume: + saved = json.loads(out.read_text()) + calib = saved["calib"] + for t in saved["ticks"]: + by_tick[t["tick"]] = t | {"rig_array": canonical(np.array(t["rig"]))} + else: + # First pass: each tick warm-started from the one before. + previous = None + for tick in ticks: + t0 = time.time() + with multiprocessing.Pool(args.workers, init_worker, (directory, skin_path, tick)) as pool: + if previous is None: + x, fx = first_tick(pool, calib, fitted, log) + calib = [float(c) for c in x[len(RIG):]] + log("calibration: " + ", ".join(f"{name} {c:.3f}" for name, c in zip(CALIB, calib))) + else: + x, fx = fit(pool, np.concatenate([previous, calib]), fitted, 0.3, 4000, previous) + x, fx = settle_facing(pool, x, fx, fitted, previous, log) + record(tick, x, fx, "", t0) + previous = by_tick[tick]["rig_array"] + + # Later rounds, for a loop: every tick held near the loop's smoothed curve. No tick depends on the + # one fitted before it, so drift can't build up, and the curve has to close. + for r in range(args.rounds if whole_loop else 0): + curve = loop_curve([by_tick[t]["rig_array"] for t in ticks], args.harmonics) + for i, tick in enumerate(ticks): + t0 = time.time() + anchor = canonical(curve[i]) + with multiprocessing.Pool(args.workers, init_worker, (directory, skin_path, tick)) as pool: + # From its own fit and from the curve: the curve can pull a tick off a ridge its own + # fit slid along. + x, fx = min((fit(pool, np.concatenate([start, calib]), fitted, 0.2, 2000, anchor) + for start in (by_tick[tick]["rig_array"], anchor)), key=lambda c: c[1]) + record(tick, x, fx, f" (round {r + 1})", t0) + print("wrote", directory / "rig.json") + + +if __name__ == "__main__": + main() diff --git a/tools/capture/model.py b/tools/capture/model.py new file mode 100644 index 0000000..77c7711 --- /dev/null +++ b/tools/capture/model.py @@ -0,0 +1,351 @@ +"""The player model and our pose maths, ported from the library so poses can be rendered in Python. + +Mirrors core's LimbBend and Bezier and the fabric mod's PoseApplier: same part boxes and pivots, same +rotation order (ModelPart's Z, then Y, then X), the same bends, and the same carrying of the head and +arms by a bent torso. check_port.py compares it against the Java classes. + +Model space is Minecraft's: pixels, Y down, the player faces -Z, and its right side is at -X. +""" +from dataclasses import dataclass + +import numpy as np + +BONES = ["root", "head", "body", "right_arm", "left_arm", "right_leg", "left_leg"] + +# Default pivots and boxes (from, size) of the vanilla player model with wide arms. +PIVOTS = { + "head": (0, 0, 0), + "body": (0, 0, 0), + "right_arm": (-5, 2, 0), + "left_arm": (5, 2, 0), + "right_leg": (-1.9, 12, 0), + "left_leg": (1.9, 12, 0), +} +BOXES = { + "head": ((-4, -8, -4), (8, 8, 8)), + "body": ((-4, 0, -2), (8, 12, 4)), + "right_arm": ((-3, -2, -2), (4, 12, 4)), + "left_arm": ((-1, -2, -2), (4, 12, 4)), + "right_leg": ((-2, 0, -2), (4, 12, 4)), + "left_leg": ((-2, 0, -2), (4, 12, 4)), +} +# The outer skin layer (hat, jacket, sleeves, pants) is this much bigger on every side. +OVERLAY = {"head": 0.5, "body": 0.25, "right_arm": 0.25, "left_arm": 0.25, "right_leg": 0.25, "left_leg": 0.25} +# Where each part's base and outer layer start in a 64x64 skin. +UV = { + "head": ((0, 0), (32, 0)), + "body": ((16, 16), (16, 32)), + "right_arm": ((40, 16), (40, 32)), + "left_arm": ((32, 48), (48, 48)), + "right_leg": ((0, 16), (0, 32)), + "left_leg": ((16, 48), (0, 48)), +} + +SHARPNESS = 0.5 +SEGMENTS = 12 +ARC_SAMPLES = 64 +ROOT_PIVOT_Y = 12.8 + + +@dataclass +class Transform: + """BoneTransform: offset in pixels, rotations and bend in radians.""" + x: float = 0 + y: float = 0 + z: float = 0 + pitch: float = 0 + yaw: float = 0 + roll: float = 0 + bend: float = 0 + axis: float = 0 + + +def rotation_zyx(pitch, yaw, roll): + """The matrix ModelPart.translateAndRotate applies: Rz(roll) Ry(yaw) Rx(pitch).""" + cx, sx = np.cos(pitch), np.sin(pitch) + cy, sy = np.cos(yaw), np.sin(yaw) + cz, sz = np.cos(roll), np.sin(roll) + rx = np.array([[1, 0, 0], [0, cx, -sx], [0, sx, cx]]) + ry = np.array([[cy, 0, sy], [0, 1, 0], [-sy, 0, cy]]) + rz = np.array([[cz, -sz, 0], [sz, cz, 0], [0, 0, 1]]) + return rz @ ry @ rx + + +def axis_angle(axis, angle): + x, y, z = axis + c, s = np.cos(angle), np.sin(angle) + k = np.array([[0, -z, y], [z, 0, -x], [-y, x, 0]]) + return np.eye(3) + s * k + (1 - c) * (k @ k) + + +def _bezier(points, t): + """Points on a cubic Bézier at each t (an array), like core's Bezier.point.""" + t = np.asarray(t, float)[:, None] + p0, p1, p2, p3 = points + u = 1 - t + return u ** 3 * p0 + 3 * u * u * t * p1 + 3 * u * t * t * p2 + t ** 3 * p3 + + +def _bezier_derivative(points, t): + t = np.asarray(t, float)[:, None] + p0, p1, p2, p3 = points + u = 1 - t + return 3 * u * u * (p1 - p0) + 6 * u * t * (p2 - p1) + 3 * t * t * (p3 - p2) + + +class LimbBend: + """Port of core's LimbBend. Vectorised over points.""" + + def __init__(self, start, end, angle, axis, sharpness=SHARPNESS, segments=SEGMENTS, anchored_at_end=False): + self.start = start + self.length = end - start + self.anchored_at_end = anchored_at_end + self.segments = segments + self.angle = angle + self.dir = np.array([np.sin(axis), 0, -np.cos(axis)]) + self.n = np.array([self.dir[2], 0, -self.dir[0]]) + self.ring = np.zeros((segments + 1, 3)) + self.ring_angle = np.zeros(segments + 1) + if self.straight: + self.ring[:, 1] = start + self.length * np.arange(segments + 1) / segments + return + self._build(self._spine(min(1, max(0, sharpness)))) + + @property + def straight(self): + return abs(self.angle) < 1e-4 + + def _spine(self, sharpness): + half = self.length / 2 + top = np.array([0, self.start, 0.0]) + joint = np.array([0, self.start + half, 0.0]) + lower = np.array([self.dir[0] * np.sin(self.angle), np.cos(self.angle), self.dir[2] * np.sin(self.angle)]) + bottom = joint + lower * half + pull = 2 / 3 + sharpness / 3 + return [top, top + (joint - top) * pull, bottom + (joint - bottom) * pull, bottom] + + def _build(self, curve): + ts = np.arange(ARC_SAMPLES + 1) / ARC_SAMPLES + pts = _bezier(curve, ts) + lengths = np.concatenate([[0], np.cumsum(np.linalg.norm(np.diff(pts, axis=0), axis=1))]) + total = lengths[-1] + scale = self.length / total + top = pts[0] + # For each ring, the arc-length sample it falls in (LimbBend's while loop), then t within it. + target = total * np.arange(self.segments + 1) / self.segments + sample = np.clip(np.searchsorted(lengths, target, side="left") - 1, 0, ARC_SAMPLES - 1) + span = lengths[sample + 1] - lengths[sample] + f = np.where(span > 0, (target - lengths[sample]) / np.where(span > 0, span, 1), 0) + t = ts[sample] + (ts[sample + 1] - ts[sample]) * f + self.ring = top + (_bezier(curve, t) - top) * scale + tangent = _bezier_derivative(curve, t) + self.ring_angle = np.arctan2(tangent @ self.dir, tangent[:, 1]) + + def _mirror(self, y): + return 2 * self.start + self.length - y + + def _rotate(self, v, a): + """Rodrigues' rotation of rows of v around n by per-row angles a.""" + n = self.n + cos, sin = np.cos(a)[:, None], np.sin(a)[:, None] + cross = np.cross(np.broadcast_to(n, v.shape), v) + dot = (v @ n)[:, None] + return v * cos + cross * sin + n * dot * (1 - cos) + + def _from_start(self, p): + along = (p[:, 1] - self.start) / self.length * self.segments + ring = np.clip(np.floor(along).astype(int), 0, self.segments - 1) + f = np.clip(along - ring, 0, 1) + extra = np.where(along <= 0, p[:, 1] - self.start, + np.where(along >= self.segments, p[:, 1] - (self.start + self.length), 0)) + f = np.where(along <= 0, 0, np.where(along >= self.segments, 1, f)) + center = self.ring[ring] + (self.ring[ring + 1] - self.ring[ring]) * f[:, None] + a = self.ring_angle[ring] + (self.ring_angle[ring + 1] - self.ring_angle[ring]) * f + local = np.stack([p[:, 0], extra, p[:, 2]], axis=1) + return self._rotate(local, a) + center + + def bend_points(self, p): + if self.anchored_at_end: + q = p.copy() + q[:, 1] = self._mirror(q[:, 1]) + out = self._from_start(q) + out[:, 1] = self._mirror(out[:, 1]) + return out + return self._from_start(p) + + def _angle_from_start(self, y): + along = np.clip((y - self.start) / self.length * self.segments, 0, self.segments) + ring = min(int(along), self.segments - 1) + return self.ring_angle[ring] + (self.ring_angle[ring + 1] - self.ring_angle[ring]) * (along - ring) + + def angle_at(self, y): + return -self._angle_from_start(self._mirror(y)) if self.anchored_at_end else self._angle_from_start(y) + + +def limb_bend(bone, t): + if t is None or t.bend == 0: + return None + if bone in ("right_arm", "left_arm"): + return LimbBend(-2, 10, t.bend, t.axis) + if bone in ("right_leg", "left_leg"): + return LimbBend(0, 12, t.bend, t.axis) + if bone == "body": + return LimbBend(0, 12, t.bend, t.axis, anchored_at_end=True) + return None + + +def textured_points(bone, skin, spacing=0.5): + """Points on a part's base and outer layer with their skin colours, in the part's own space. + + Follows ModelPart.Cube: each face's corners and UV rectangle, and Polygon's corner-to-UV order. + Outer layer points are only kept where the skin has pixels there. + """ + (fx, fy, fz), (dx, dy, dz) = BOXES[bone] + pts, cols = [], [] + for layer, (u, v) in enumerate(UV[bone]): + g = OVERLAY[bone] if layer else 0 + f, gg, h = fx - g, fy - g, fz - g + i, j, k = fx + dx + g, fy + dy + g, fz + dz + g + c = [(f, gg, h), (i, gg, h), (i, j, h), (f, j, h), (f, gg, k), (i, gg, k), (i, j, k), (f, j, k)] + l, m, n, o = u, u + dz, u + dz + dx, u + dz + dx + dx + pp, q = u + dz + dx + dz, u + dz + dx + dz + dx + r, ss, t = v, v + dz, v + dz + dy + faces = [ # corners, (u1, v1, u2, v2) + ((5, 4, 0, 1), (m, r, n, ss)), # DOWN: minY, the top in model space + ((2, 3, 7, 6), (n, ss, o, r)), # UP: maxY, the bottom + ((0, 4, 7, 3), (l, ss, m, t)), # WEST: -X, the player's right + ((1, 0, 3, 2), (m, ss, n, t)), # NORTH: -Z, the front + ((5, 1, 2, 6), (n, ss, pp, t)), # EAST: +X, the player's left + ((4, 5, 6, 7), (pp, ss, q, t)), # SOUTH: +Z, the back + ] + for corners, (u1, v1, u2, v2) in faces: + a, b, _, d = (np.array(c[x], float) for x in corners) + ns = max(2, int(np.ceil(np.linalg.norm(b - a) / spacing)) + 1) + nt = max(2, int(np.ceil(np.linalg.norm(d - a) / spacing)) + 1) + # Sample cell centres so every point falls inside one texture pixel. + sv, tv = np.meshgrid((np.arange(ns) + 0.5) / ns, (np.arange(nt) + 0.5) / nt, indexing="ij") + sv, tv = sv.ravel(), tv.ravel() + face = a + sv[:, None] * (b - a) + tv[:, None] * (d - a) + tu = np.clip(np.floor(u2 + sv * (u1 - u2)).astype(int), min(u1, u2), max(u1, u2) - 1) + tvv = np.clip(np.floor(v1 + tv * (v2 - v1)).astype(int), min(v1, v2), max(v1, v2) - 1) + rgba = skin[tvv, tu] + keep = rgba[:, 3] > 0 if layer else np.ones(len(face), bool) + pts.append(face[keep]) + cols.append(rgba[keep, :3]) + return np.concatenate(pts), np.concatenate(cols).astype(float) / 255 + + +def surface_points(bone, spacing=0.5, overlay=True): + """Points on the surface of a part's box (outer layer included), in the part's own space.""" + (fx, fy, fz), (sx, sy, sz) = BOXES[bone] + g = OVERLAY[bone] if overlay else 0 + lo = np.array([fx - g, fy - g, fz - g]) + hi = np.array([fx + sx + g, fy + sy + g, fz + sz + g]) + axes = [np.linspace(lo[i], hi[i], max(2, int(np.ceil((hi[i] - lo[i]) / spacing)) + 1)) for i in range(3)] + pts = [] + for fixed in range(3): + a, b = [i for i in range(3) if i != fixed] + u, v = np.meshgrid(axes[a], axes[b], indexing="ij") + for side in (lo[fixed], hi[fixed]): + face = np.empty((u.size, 3)) + face[:, a], face[:, b], face[:, fixed] = u.ravel(), v.ravel(), side + pts.append(face) + return np.unique(np.round(np.concatenate(pts), 4), axis=0) + + +class Model: + """Poses the player's parts. pose() returns model-space points for each part. + + With a skin (an RGBA array), the points carry the skin's colours in self.colors, and the outer layer + only exists where the skin has pixels. Without one, parts are plain boxes including the outer layer. + """ + + def __init__(self, spacing=0.5, skin=None): + if skin is None: + self.local = {bone: surface_points(bone, spacing) for bone in PIVOTS} + self.colors = None + else: + textured = {bone: textured_points(bone, skin, spacing) for bone in PIVOTS} + self.local = {bone: pts for bone, (pts, _) in textured.items()} + self.colors = np.concatenate([cols for _, cols in textured.values()]) + + def part_frames(self, pose): + """Each part's pivot, rotation and bend after PoseApplier.apply, in root space.""" + frames = {} + for bone in PIVOTS: + t = pose.get(bone) + pivot = np.array(PIVOTS[bone], float) + rot = np.eye(3) + if t is not None: + pivot = pivot + [t.x, t.y, t.z] + rot = rotation_zyx(t.pitch, t.yaw, t.roll) + frames[bone] = [pivot, rot, limb_bend(bone, t)] + + body_pivot, body_rot, body_bend = frames["body"] + if body_bend is not None and not body_bend.straight: + for bone in ("head", "right_arm", "left_arm"): + pivot, rot, bend = frames[bone] + local = body_rot.T @ (pivot - body_pivot) + moved = body_bend.bend_points(local[None, :])[0] + turn = axis_angle(body_bend.n, body_bend.angle_at(local[1])) + frames[bone] = [body_pivot + body_rot @ moved, body_rot @ turn @ body_rot.T @ rot, bend] + return frames + + def pose(self, pose): + frames = self.part_frames(pose) + root = pose.get("root") + if root is not None: + r_rot = rotation_zyx(root.pitch, root.yaw, root.roll) + pivot = np.array([0, ROOT_PIVOT_Y, 0]) + r_pos = np.array([root.x, root.y, root.z]) + pivot - r_rot @ pivot + else: + r_rot, r_pos = np.eye(3), np.zeros(3) + out = {} + for bone, (pivot, rot, bend) in frames.items(): + p = self.local[bone] + if bend is not None and not bend.straight: + p = bend.bend_points(p) + out[bone] = (p @ rot.T + pivot) @ r_rot.T + r_pos + return out + + +# --- The rig the fitter works in ------------------------------------------------------------------- +# +# Moving every pivot freely lets parts drift apart, so the fitter uses a jointed rig instead and +# converts it to BoneTransforms: the legs hang from the hips, the torso's hip end sits on the hips, +# and the head and arms hang from the torso (PoseApplier then carries them along its bend). + +RIG = ( + ["root_x", "root_y", "root_z", "root_pitch", "root_yaw", "root_roll", "hip_x", "hip_y", "hip_z"] + + [f"body_{c}" for c in ("pitch", "yaw", "roll", "bend", "axis")] + + [f"head_{c}" for c in ("pitch", "yaw", "roll")] + + [f"{limb}_{c}" for limb in ("right_arm", "left_arm", "right_leg", "left_leg") + for c in ("pitch", "yaw", "roll", "bend", "axis")] +) +INDEX = {name: i for i, name in enumerate(RIG)} + + +def rig_to_pose(v): + g = lambda name: float(v[INDEX[name]]) + hip = np.array([g("hip_x"), g("hip_y"), g("hip_z")]) + body_rot = rotation_zyx(g("body_pitch"), g("body_yaw"), g("body_roll")) + # The torso's bottom middle (0, 12, 0) in its own space stays on the hips; a bend keeps it fixed. + hips_rest = np.array([0, 12, 0.0]) + body_pivot = hips_rest + hip - body_rot @ hips_rest + pose = { + "root": Transform(g("root_x"), g("root_y"), g("root_z"), g("root_pitch"), g("root_yaw"), g("root_roll")), + "body": Transform(*body_pivot, g("body_pitch"), g("body_yaw"), g("body_roll"), g("body_bend"), g("body_axis")), + "head": Transform(*body_pivot, g("head_pitch"), g("head_yaw"), g("head_roll")), + } + for arm in ("right_arm", "left_arm"): + rest = np.array(PIVOTS[arm], float) + # Where the shoulder sits on the (unbent) tipped torso; PoseApplier adds the bend's carry. + offset = body_pivot + body_rot @ rest - rest + pose[arm] = Transform(*offset, *(g(f"{arm}_{c}") for c in ("pitch", "yaw", "roll", "bend", "axis"))) + for leg in ("right_leg", "left_leg"): + pose[leg] = Transform(*hip, *(g(f"{leg}_{c}") for c in ("pitch", "yaw", "roll", "bend", "axis"))) + return pose + + +def rest_rig(): + return np.zeros(len(RIG)) diff --git a/tools/capture/overlay.py b/tools/capture/overlay.py new file mode 100644 index 0000000..83b8249 --- /dev/null +++ b/tools/capture/overlay.py @@ -0,0 +1,63 @@ +"""Draws the fitted pose's outline over the captured frames, to see where the fit is off. + + .venv/bin/python overlay.py [--ticks 0:13:3] [--views y000,y090,y045,top000] + +Writes frames//overlay.png: rows are views, columns ticks. Red is where only our model is, blue +where only theirs is, and grey where both are. +""" +import argparse +import json +from pathlib import Path + +import numpy as np +from PIL import Image + +from model import Model, rig_to_pose +from render import Camera, render_masks + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("name") + parser.add_argument("--ticks", default="0:13:3") + parser.add_argument("--views", default="y000,y090,y045,top000") + args = parser.parse_args() + + base = Path("frames") / args.name + meta = json.loads((base / "views.json").read_text()) + data = json.loads((base / "rig.json").read_text()) + by_tick = {t["tick"]: np.array(t["rig"]) for t in data["ticks"]} + ticks = [t for t in range(*map(int, args.ticks.split(":"))) if t in by_tick] + views = args.views.split(",") + scale = 2 + cams = {v: Camera(meta["views"][v], meta["size"], meta["aspect"], scale) for v in views} + model = Model(spacing=0.25) + + tiles = [] + for v in views: + row = [] + for t in ticks: + pts = np.concatenate(list(model.pose(rig_to_pose(by_tick[t])).values())) + ours = render_masks(pts, data["calib"], {v: cams[v]})[v] + frame = Image.open(base / v / f"{t:04d}.png").reduce(scale) + theirs = np.asarray(frame)[:, :, 3] > 127 + img = np.full(ours.shape + (3,), 30, np.uint8) + img[theirs & ours] = (150, 150, 150) + img[ours & ~theirs] = (230, 60, 60) + img[theirs & ~ours] = (60, 110, 240) + row.append(img) + tiles.append(row) + + both = np.zeros(tiles[0][0].shape[:2], bool) + for row in tiles: + for img in row: + both |= img.sum(axis=2) > 90 + ys, xs = np.nonzero(both) + y0, y1, x0, x1 = max(ys.min() - 4, 0), ys.max() + 4, max(xs.min() - 4, 0), xs.max() + 4 + sheet = np.concatenate([np.concatenate([img[y0:y1, x0:x1] for img in row], axis=1) for row in tiles], axis=0) + Image.fromarray(sheet).save(base / "overlay.png") + print("wrote", base / "overlay.png", "ticks", ticks) + + +if __name__ == "__main__": + main() diff --git a/tools/capture/render.py b/tools/capture/render.py new file mode 100644 index 0000000..20aa95a --- /dev/null +++ b/tools/capture/render.py @@ -0,0 +1,174 @@ +"""Renders model points through the cameras capture.py used, and scores them against captured masks.""" +import json +from pathlib import Path + +import numpy as np +from PIL import Image +from scipy.ndimage import distance_transform_edt + +# Model space (Y down, facing -Z) to the viewer's world (Y up, facing +Z): Minecraft's scale(-1, -1, 1) +# after turning the player 180 degrees to face south. +FLIP = np.diag([1.0, -1.0, -1.0]) + + +class Camera: + """A three.js PerspectiveCamera placed with lookAt, as viewer.js places it.""" + + def __init__(self, view, size, aspect, scale): + yaw, pitch = np.radians(view["yaw"]), np.radians(view["pitch"]) + target = np.array(view["target"], float) + d = view["distance"] + self.position = target + d * np.array([np.cos(pitch) * np.sin(yaw), np.sin(pitch), np.cos(pitch) * np.cos(yaw)]) + z = self.position - target + z /= np.linalg.norm(z) + x = np.cross([0, 1, 0], z) + x /= np.linalg.norm(x) + y = np.cross(z, x) + self.rotation = np.stack([x, y, z]) # world to camera + self.f = 1 / np.tan(np.radians(view["fov"]) / 2) + self.aspect = aspect + self.width, self.height = size[0] // scale, size[1] // scale + + def project(self, world): + c = (world - self.position) @ self.rotation.T + depth = -c[:, 2] + nx = self.f / self.aspect * c[:, 0] / depth + ny = self.f * c[:, 1] / depth + return np.stack([(nx + 1) / 2 * self.width, (1 - ny) / 2 * self.height], axis=1) + + +def to_world(points, calib): + """calib = (scale, tx, ty, tz): where the viewer puts the model.""" + return calib[0] * points @ FLIP.T + np.asarray(calib[1:4]) + + +def splat(px, width, height): + """A mask covering every projected point, 2x2 pixels each so the surface has no holes.""" + mask = np.zeros((height, width), bool) + x = np.floor(px[:, 0] - 0.5).astype(int) + y = np.floor(px[:, 1] - 0.5).astype(int) + for dx in (0, 1): + for dy in (0, 1): + xs, ys = x + dx, y + dy + ok = (xs >= 0) & (xs < width) & (ys >= 0) & (ys < height) + mask[ys[ok], xs[ok]] = True + return mask + + +def color_features(rgb): + """Lighting-tolerant colour: hue and saturation as opponent channels divided by brightness, plus + log brightness. Shading scales all three channels, which leaves the first two alone.""" + lum = rgb.mean(axis=-1) + o1 = rgb[..., 0] - rgb[..., 1] + o2 = (rgb[..., 0] + rgb[..., 1]) / 2 - rgb[..., 2] + return np.stack([o1 / (lum + 0.05), o2 / (lum + 0.05), np.log(lum + 0.05)], axis=-1) + + +class Target: + """One captured view of one tick, downscaled, with the distance maps the score needs.""" + + def __init__(self, mask, rgb=None): + self.mask = mask + self.outside = distance_transform_edt(~mask) + self.ys, self.xs = np.nonzero(mask) + self.features = None if rgb is None else color_features(rgb) + + +def downscale(alpha, scale): + h, w = alpha.shape + blocks = alpha[: h // scale * scale, : w // scale * scale].reshape(h // scale, scale, w // scale, scale) + return blocks.mean(axis=(1, 3)) > 127 + + +def downscale_rgb(rgba, scale): + """Mean colour of the covered pixels in each block (0-1).""" + h, w = rgba.shape[:2] + blocks = rgba[: h // scale * scale, : w // scale * scale].astype(float) + blocks = blocks.reshape(h // scale, scale, w // scale, scale, 4) + alpha = blocks[..., 3:] / 255 + return (blocks[..., :3] / 255 * alpha).sum(axis=(1, 3)) / np.maximum(alpha.sum(axis=(1, 3)), 1e-6) + + +class Capture: + """The frames and cameras of one capture.""" + + def __init__(self, directory, scale=4, views=None): + self.dir = Path(directory) + self.meta = json.loads((self.dir / "views.json").read_text()) + self.scale = scale + self.names = views or list(self.meta["views"]) + self.cameras = {n: Camera(self.meta["views"][n], self.meta["size"], self.meta["aspect"], scale) for n in self.names} + + def targets(self, tick): + out = {} + for n in self.names: + rgba = np.asarray(Image.open(self.dir / n / f"{tick:04d}.png")) + out[n] = Target(downscale(rgba[:, :, 3], self.scale), downscale_rgb(rgba, self.scale)) + return out + + +# How much a unit of colour difference counts against a pixel of outline distance, and how much +# brightness counts next to hue and saturation. +COLOR_WEIGHT = 0.5 +BRIGHTNESS_WEIGHT = 0.3 + + +def score(points, calib, cameras, targets, per_view=False, colors=None, color_views=None, one_way=False, + color_weight=None): + """Mean outline distance in pixels, both ways (our points outside theirs and theirs outside ours), + plus, with colours, how different the colours are where both outlines cover a pixel. + + one_way only counts our points outside theirs, for fitting some of the parts: the rest of their + outline is still unexplained, and shouldn't pull the parts being fitted towards it.""" + world = to_world(points, calib) + total = {} + for name, cam in cameras.items(): + t = targets[name] + px = cam.project(world) + xi = np.clip(px[:, 0].astype(int), 0, cam.width - 1) + yi = np.clip(px[:, 1].astype(int), 0, cam.height - 1) + ours_out = t.outside[yi, xi].mean() + if colors is not None and t.features is not None and (color_views is None or name in color_views): + depth = -((world - cam.position) @ cam.rotation.T)[:, 2] + image, mask = splat_color(px, depth, colors, cam.width, cam.height) + both = mask & t.mask + diff = np.abs(color_features(image[both]) - t.features[both]) + color = (diff[:, 0] + diff[:, 1] + BRIGHTNESS_WEIGHT * diff[:, 2]).mean() if both.any() else 0 + else: + mask = splat(px, cam.width, cam.height) + color = 0 + theirs_out = 0 if one_way or not len(t.ys) else distance_transform_edt(~mask)[t.ys, t.xs].mean() + total[name] = ours_out + theirs_out + (COLOR_WEIGHT if color_weight is None else color_weight) * color + return total if per_view else sum(total.values()) / len(total) + + +def render_masks(points, calib, cameras): + world = to_world(points, calib) + return {n: splat(c.project(world), c.width, c.height) for n, c in cameras.items()} + + +def splat_color(px, depth, colors, width, height): + """Colour image of the nearest point in each pixel (2x2 pixels per point), and its coverage mask.""" + order = np.argsort(-depth) # far to near, so nearer points are written last and win + x = np.floor(px[order, 0] - 0.5).astype(int) + y = np.floor(px[order, 1] - 0.5).astype(int) + # All four pixels of every point in one write; a write per offset would let far points overwrite + # pixels near points filled at another offset. + xs = np.stack([x, x + 1, x, x + 1], axis=1).ravel() + ys = np.stack([y, y, y + 1, y + 1], axis=1).ravel() + cs = np.repeat(colors[order], 4, axis=0) + ok = (xs >= 0) & (xs < width) & (ys >= 0) & (ys < height) + image = np.zeros((height, width, 3)) + mask = np.zeros((height, width), bool) + image[ys[ok], xs[ok]] = cs[ok] + mask[ys[ok], xs[ok]] = True + return image, mask + + +def render_colors(points, colors, calib, cameras): + world = to_world(points, calib) + out = {} + for n, c in cameras.items(): + depth = -((world - c.position) @ c.rotation.T)[:, 2] + out[n] = splat_color(c.project(world), depth, colors, c.width, c.height) + return out diff --git a/tools/capture/requirements.txt b/tools/capture/requirements.txt new file mode 100644 index 0000000..9f27a86 --- /dev/null +++ b/tools/capture/requirements.txt @@ -0,0 +1,5 @@ +cma==4.5.0 +numpy==2.5.3 +pillow==12.3.0 +playwright==1.63.0 +scipy==1.18.1 diff --git a/tools/capture/sheet.py b/tools/capture/sheet.py new file mode 100644 index 0000000..b917e2b --- /dev/null +++ b/tools/capture/sheet.py @@ -0,0 +1,20 @@ +"""Tiles frames into a contact sheet: rows are views, columns are ticks, cropped to the player.""" +import sys +from pathlib import Path + +import numpy as np +from PIL import Image + +name, views, ticks = sys.argv[1], sys.argv[2].split(","), range(*map(int, sys.argv[3].split(":"))) +out = sys.argv[4] if len(sys.argv) > 4 else "sheet.png" +frames = [[Image.open(Path("frames") / name / v / f"{t:04d}.png") for t in ticks] for v in views] +alpha = np.maximum.reduce([np.asarray(f)[:, :, 3] for row in frames for f in row]) +ys, xs = np.nonzero(alpha) +box = (xs.min() - 4, ys.min() - 4, xs.max() + 4, ys.max() + 4) +w, h = box[2] - box[0], box[3] - box[1] +sheet = Image.new("RGBA", (w * len(ticks), h * len(views)), (40, 40, 52, 255)) +for r, row in enumerate(frames): + for c, f in enumerate(row): + sheet.alpha_composite(f.crop(box), (c * w, r * h)) +sheet.save(out) +print(sheet.size) diff --git a/tools/capture/skin.py b/tools/capture/skin.py new file mode 100644 index 0000000..3f2c889 --- /dev/null +++ b/tools/capture/skin.py @@ -0,0 +1,29 @@ +"""Fetches a player's skin texture from Mojang, the same one the store viewer shows for that username. + + .venv/bin/python skin.py [out.png] +""" +import base64 +import json +import subprocess +import sys + + +def get(url): + # curl uses the system's certificate store, which python.org's Python doesn't. + return subprocess.run(["curl", "-sSf", url], capture_output=True, check=True).stdout + + +def fetch(username): + """Returns (png bytes, slim arms).""" + uuid = json.loads(get(f"https://api.mojang.com/users/profiles/minecraft/{username}"))["id"] + profile = json.loads(get(f"https://sessionserver.mojang.com/session/minecraft/profile/{uuid}")) + prop = next(p for p in profile["properties"] if p["name"] == "textures") + skin = json.loads(base64.b64decode(prop["value"]))["textures"]["SKIN"] + return get(skin["url"]), skin.get("metadata", {}).get("model") == "slim" + + +if __name__ == "__main__": + png, slim = fetch(sys.argv[1]) + out = sys.argv[2] if len(sys.argv) > 2 else f"skins/{sys.argv[1]}.png" + open(out, "wb").write(png) + print(out, "slim" if slim else "wide") diff --git a/tools/capture/symmetry.py b/tools/capture/symmetry.py new file mode 100644 index 0000000..bb2ba9c --- /dev/null +++ b/tools/capture/symmetry.py @@ -0,0 +1,110 @@ +"""Left/right symmetry of a measured loop: how lopsided it is, and a symmetric version of it. + + .venv/bin/python symmetry.py [--rig path.json] + +A mirror swaps the left and right limbs and flips every sideways channel (the hips' x, yaws, rolls and +bend directions). A pose that sways once per loop should be the mirror of itself half a loop later; +one with no sideways motion, the mirror of itself at the same moment. Whatever breaks that is +lopsidedness the fit picked up (one leg bent more, one side swinging further), not the dance. + +symmetrize() averages each tick with the mirror of its partner, which keeps the sway and removes the +rest. fit.py --symmetric uses it. +""" +import argparse +import json +from pathlib import Path + +import numpy as np + +from model import INDEX, RIG + +SIDES = {"right_arm": "left_arm", "left_arm": "right_arm", "right_leg": "left_leg", "left_leg": "right_leg"} +ANGLES = np.array([not name.endswith(("_x", "_y", "_z")) for name in RIG]) + + +def wrap(a): + return (a + np.pi) % (2 * np.pi) - np.pi + + +def mirror(rig): + """The pose reflected through the player's middle (x -> -x). Rotations about x keep their angle and + those about y and z flip (S Rz Ry Rx S with S = diag(-1, 1, 1)); a bend direction, measured around + the limb from its front, flips too.""" + out = np.empty_like(rig) + for i, name in enumerate(RIG): + part, _, channel = name.rpartition("_") + source = next((f"{SIDES[p]}{name[len(p):]}" for p in SIDES if name.startswith(p)), name) + value = rig[INDEX[source]] + out[i] = -value if channel in ("x", "yaw", "roll", "axis") else value + return out + + +def shifted(rigs, fraction): + """Each channel of a loop evaluated `fraction` of a loop later, between samples too, through its + Fourier series. Angles that go round whole turns in a loop keep doing so.""" + rigs = np.asarray(rigs, float) + n = len(rigs) + out = np.empty_like(rigs) + k = np.arange(n // 2 + 1) + for i in range(rigs.shape[1]): + values = np.unwrap(rigs[:, i]) if ANGLES[i] else rigs[:, i] + turns = 0.0 + if ANGLES[i]: + turns = values[-1] + wrap(values[0] - values[-1]) - values[0] + ramp = turns * np.arange(n) / n + spectrum = np.fft.rfft(values - ramp) * np.exp(2j * np.pi * k * fraction) + out[:, i] = np.fft.irfft(spectrum, n) + ramp + turns * fraction + return out + + +def partners(rigs, sway): + """The mirrored partner of every tick: half a loop on for a pose that sways once per loop.""" + rigs = np.asarray(rigs, float) + source = shifted(rigs, 0.5) if sway else rigs + return np.array([mirror(r) for r in source]) + + +def difference(a, b): + d = a - b + return np.where(ANGLES, wrap(d), d) + + +def symmetrize(rigs, sway): + """Each tick averaged with its mirrored partner (the short way round for angles).""" + rigs = np.asarray(rigs, float) + return rigs - difference(rigs, partners(rigs, sway)) / 2 + + +def lopsidedness(rigs, sway): + """Per channel: how far each tick is from its mirrored partner, as (mean, rms) over the loop. The + mean is a steady lean to one side; the rms includes one side swinging further than the other.""" + d = difference(np.asarray(rigs, float), partners(rigs, sway)) + return d.mean(axis=0), np.sqrt((d ** 2).mean(axis=0)) + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("name") + parser.add_argument("--rig") + args = parser.parse_args() + path = Path(args.rig) if args.rig else Path("frames") / args.name / "rig.json" + rigs = np.array([t["rig"] for t in json.loads(path.read_text())["ticks"]]) + + # Which pairing fits: a sway once per loop, or none. + for sway in (True, False): + _, rms = lopsidedness(rigs, sway) + angle_rms = np.degrees(np.sqrt((rms[ANGLES] ** 2).mean())) + offset_rms = np.sqrt((rms[~ANGLES] ** 2).mean()) + print(f"{'sway once per loop' if sway else 'no sway':18s}: rms {angle_rms:.1f} deg, {offset_rms:.2f} px") + + sway = True + mean, rms = lopsidedness(rigs, sway) + print("\nMost lopsided channels (tick vs its mirrored partner half a loop on):") + order = np.argsort(-np.where(ANGLES, np.degrees(rms), rms * 5)) + for i in order[:14]: + unit, f = ("deg", np.degrees) if ANGLES[i] else ("px", lambda v: v) + print(f" {RIG[i]:18s} steady {f(mean[i]):+6.1f} {unit}, rms {f(rms[i]):5.1f} {unit}") + + +if __name__ == "__main__": + main() diff --git a/tools/capture/synth_test.py b/tools/capture/synth_test.py new file mode 100644 index 0000000..a13c019 --- /dev/null +++ b/tools/capture/synth_test.py @@ -0,0 +1,85 @@ +"""End-to-end check of measure.py on frames with a known answer: our own twerk test animation, rendered +by model.py as if it were a capture, then measured like one. + + .venv/bin/python synth_test.py make [ticks] # writes frames/synth + .venv/bin/python measure.py synth --ticks 0:4 + .venv/bin/python synth_test.py check +""" +import json +import sys +from pathlib import Path + +import numpy as np +from PIL import Image +from scipy.spatial import cKDTree + +from model import INDEX, Model, rest_rig, rig_to_pose +from render import Camera, render_colors + +# Where the fake viewer puts the model: Minecraft's player scale, feet 25 units below the origin. +CALIB = (0.9375, 0, -1.5, 0) +SKIN = "Steve" +OUT = Path("frames/synth") + + +def twerk(t): + """TestAnimations.twerk, written in rig terms.""" + up = 0.5 * (1 - np.cos(t / 8 * 2 * np.pi)) + thigh, knee, splay, tilt, arch = -1.1 + 0.15 * up, 1.75 - 0.3 * up, 0.55, 1.4, 0.4 + 0.3 * up + drop = 12 - 6 * (np.cos(thigh) + np.cos(thigh + knee)) * np.cos(splay) + back = -6 * (np.sin(thigh) + np.sin(thigh + knee)) + chest = tilt - arch + v = rest_rig() + s = lambda name, value: v.__setitem__(INDEX[name], value) + s("hip_y", drop), s("hip_z", back) + s("body_pitch", tilt), s("body_bend", arch), s("body_axis", np.pi) + s("head_pitch", tilt - chest * 0.5) + for arm, roll in (("right_arm", 0.3), ("left_arm", -0.3)): + s(f"{arm}_pitch", tilt - chest * 1.1), s(f"{arm}_roll", roll), s(f"{arm}_bend", 0.5) + for leg, sign in (("right_leg", 1), ("left_leg", -1)): + s(f"{leg}_pitch", thigh), s(f"{leg}_roll", sign * splay), s(f"{leg}_bend", knee) + s(f"{leg}_axis", np.pi - sign * 0.35) + return v + + +def skin(): + return np.asarray(Image.open(f"skins/{SKIN}.png").convert("RGBA")) + + +def make(source, ticks): + meta = json.loads((Path("frames") / source / "views.json").read_text()) + model = Model(spacing=0.08, skin=skin()) + cams = {n: Camera(v, meta["size"], meta["aspect"], 1) for n, v in meta["views"].items()} + for tick in range(ticks): + pts = np.concatenate(list(model.pose(rig_to_pose(twerk(tick))).values())) + for name, (image, mask) in render_colors(pts, model.colors, CALIB, cams).items(): + (OUT / name).mkdir(parents=True, exist_ok=True) + rgba = np.concatenate([image * 255, mask[..., None] * 255], axis=-1).astype(np.uint8) + Image.fromarray(rgba).save(OUT / name / f"{tick:04d}.png") + meta |= {"url": "synthetic twerk", "skin": SKIN, "ticks": ticks, "loop_ticks": None} + (OUT / "views.json").write_text(json.dumps(meta, indent=2)) + print(f"wrote {ticks} ticks to {OUT}") + + +def check(): + data = json.loads((OUT / "rig.json").read_text()) + print("calibration:", np.round(data["calib"], 3), "true:", CALIB) + model = Model(skin=skin()) + for entry in data["ticks"]: + truth, got = model.pose(rig_to_pose(twerk(entry["tick"]))), model.pose(rig_to_pose(np.array(entry["rig"]))) + # The fitted pose sits in the fitted calibration's frame; compare where the viewer shows them. + to = lambda p, c: c[0] * p + np.asarray(c[1:]) * [1, -1, -1] + parts = [] + for part in truth: + a, b = to(truth[part], CALIB), to(got[part], data["calib"]) + parts.append((part, np.linalg.norm(a - b, axis=1).mean(), cKDTree(b).query(a)[0].mean())) + print(f"tick {entry['tick']}: score {entry['score']:.3f} " + + " ".join(f"{p} {d:.2f}/{s:.2f}" for p, d, s in parts)) + print("(per part: mean point-for-point / shape-only distance, in model pixels)") + + +if __name__ == "__main__": + if sys.argv[1] == "make": + make(sys.argv[2], int(sys.argv[3]) if len(sys.argv) > 3 else 4) + else: + check() diff --git a/tools/capture/viewer.js b/tools/capture/viewer.js new file mode 100644 index 0000000..46a38ad --- /dev/null +++ b/tools/capture/viewer.js @@ -0,0 +1,88 @@ +// Injected before the page loads. It puts the page on a clock we step by hand, and uses three.js's +// devtools hook to find the renderer and the camera it draws with. Only the clock and the camera are +// touched: the frames are measured from pixels, never from the scene's animation data. +(() => { + let now = 0; + let callbacks = []; + let nextId = 1; + + const realDateNow = Date.now.bind(Date); + const epoch = realDateNow(); + performance.now = () => now; + Date.now = () => epoch + now; + window.requestAnimationFrame = (cb) => { + const id = nextId++; + callbacks.push({ id, cb }); + return id; + }; + window.cancelAnimationFrame = (id) => { + callbacks = callbacks.filter((c) => c.id !== id); + }; + + // The camera placement to force on every render: a point the camera orbits, its distance, and the + // yaw/pitch it looks from (degrees; yaw 0 looks at the model's front). + let view = null; + let lastFrame = null; + + const capture = { + renderers: [], + camera: null, + /** Advances the clock by ms and runs one animation frame. Returns the frame as a PNG data URL. */ + step(ms, grab) { + now += ms; + lastFrame = null; + const due = callbacks; + callbacks = []; + for (const { cb } of due) { + try { cb(now); } catch (e) { console.error(e); } + } + return grab ? lastFrame : null; + }, + time() { return now; }, + setView(v) { view = v; }, + camera_info() { + const cam = capture.camera; + return cam && { fov: cam.fov, aspect: cam.aspect, near: cam.near, far: cam.far, + position: cam.position.toArray(), quaternion: cam.quaternion.toArray() }; + }, + }; + window.__capture = capture; + + function place(camera) { + const toRad = Math.PI / 180; + const yaw = view.yaw * toRad; + const pitch = view.pitch * toRad; + const [tx, ty, tz] = view.target; + camera.position.set( + tx + view.distance * Math.cos(pitch) * Math.sin(yaw), + ty + view.distance * Math.sin(pitch), + tz + view.distance * Math.cos(pitch) * Math.cos(yaw)); + camera.up.set(0, 1, 0); + camera.lookAt(tx, ty, tz); + if (view.fov) { + camera.fov = view.fov; + } + camera.updateProjectionMatrix(); + camera.updateMatrixWorld(true); + } + + const hook = new EventTarget(); + hook.addEventListener('observe', (event) => { + const obj = event.detail; + if (obj && obj.isWebGLRenderer && !capture.renderers.includes(obj)) { + capture.renderers.push(obj); + const render = obj.render.bind(obj); + obj.render = (scene, camera) => { + capture.camera = camera; + if (view) { + place(camera); + } + const result = render(scene, camera); + // Read the drawing buffer in the same task as the draw, before the browser clears it. + lastFrame = obj.domElement.toDataURL('image/png'); + return result; + }; + } + }); + window.__THREE_DEVTOOLS__ = hook; +})();