Add a tool that measures emotes from a store's 3D preview

tools/capture drives a store page's skin viewer in headless Firefox, stepping its clock one tick at
a time and taking every tick from ten fixed cameras. It then fits our own rig to the frames by
rendering the model (a Python port of LimbBend and PoseApplier) and matching outlines and colours,
and writes the result as Emotecraft JSON. It measures pixels only and never reads the page's
animation data. The README covers the steps, the checks and the limits.

Co-Authored-By: Claude Opus 5.5 <[email protected]>
This commit is contained in:
2026-09-29 16:55:08 +02:00
co-authored by claude
parent 7f8de39253
commit df9a2587e6
19 changed files with 1785 additions and 0 deletions
+140
View File
@@ -0,0 +1,140 @@
"""Captures an emote from a store page's 3D viewer, one frame per tick from several fixed cameras.
.venv/bin/python capture.py <store url> <name> [--seconds 6]
Writes frames/<name>/<view>/<tick>.png (RGBA, transparent background) and frames/<name>/views.json,
which records each camera exactly so measure.py can render the same views.
The page runs on a clock that viewer.js steps by hand, so every view of a tick shows the same moment
and nothing is missed however fast the emote moves.
"""
import argparse
import base64
import io
import json
from pathlib import Path
import numpy as np
from PIL import Image
from playwright.sync_api import sync_playwright
TICK_MS = 50
# Where the cameras look (viewer world units) and from how far. The model stands about one unit per
# model pixel, with its middle about 12 units below the viewer's origin.
TARGET = [0, -12, 0]
DISTANCE = 50
FOV = 50
VIEWS = {f"y{yaw:03d}": (yaw, 0) for yaw in range(0, 360, 45)}
VIEWS |= {"top000": (0, 50), "top090": (90, 50)}
def step(page, ms, grab=False):
url = page.evaluate(f"window.__capture.step({ms}, {'true' if grab else 'false'})")
if not grab:
return None
return Image.open(io.BytesIO(base64.b64decode(url.split(",", 1)[1]))).convert("RGBA")
def warm_up(page, frames):
for _ in range(frames):
step(page, 16)
page.wait_for_timeout(15)
def set_view(page, yaw, pitch):
view = {"target": TARGET, "distance": DISTANCE, "yaw": yaw, "pitch": pitch, "fov": FOV}
page.evaluate(f"window.__capture.setView({json.dumps(view)})")
return view
def choose_skin(page, skin):
"""Puts a skin we have the texture of on the model, so measure.py can match colours too."""
page.evaluate("[...document.querySelectorAll('button')].find(b => b.innerText.includes(\"'s skin\")).click()")
warm_up(page, 40)
page.get_by_text("Change preview skin").last.click(force=True)
warm_up(page, 60)
preset = page.get_by_role("button", name=skin, exact=True)
if preset.count():
preset.first.click(force=True)
else:
page.get_by_placeholder("Enter a Minecraft username").fill(skin)
page.keyboard.press("Enter")
warm_up(page, 150)
page.keyboard.press("Escape")
def loop_ticks(masks):
"""The loop length in ticks: the shift that best maps the front view's masks onto themselves."""
best, best_err = None, None
for period in range(8, len(masks) // 2):
err = np.mean([np.mean(masks[i] != masks[i + period]) for i in range(len(masks) - period)])
if best_err is None or err < best_err:
best, best_err = period, err
return best, best_err
def main():
parser = argparse.ArgumentParser()
parser.add_argument("url")
parser.add_argument("name")
parser.add_argument("--seconds", type=float, default=6)
parser.add_argument("--skin", default="Steve",
help="a preset from the viewer's skin dialog (Steve, Alex, ...) or a Minecraft username")
args = parser.parse_args()
out = Path("frames") / args.name
ticks = round(args.seconds * 1000 / TICK_MS)
with sync_playwright() as p:
browser = p.firefox.launch()
page = browser.new_page(viewport={"width": 1280, "height": 900})
page.add_init_script(path=str(Path(__file__).with_name("viewer.js")))
page.goto(args.url, wait_until="domcontentloaded", timeout=60000)
warm_up(page, 200)
# The toggle between the video and the 3D viewer sits in the preview's top right corner.
page.mouse.click(794, 177)
page.wait_for_selector("#skin-viewer", timeout=30000)
warm_up(page, 100)
choose_skin(page, args.skin)
warm_up(page, 200)
cameras = {}
for name, (yaw, pitch) in VIEWS.items():
(out / name).mkdir(parents=True, exist_ok=True)
cameras[name] = set_view(page, yaw, pitch)
info = page.evaluate("window.__capture.camera_info()")
start = page.evaluate("window.__capture.time()")
front_masks = []
for tick in range(ticks):
for i, (name, (yaw, pitch)) in enumerate(VIEWS.items()):
set_view(page, yaw, pitch)
# The first view advances the clock by a tick; the rest redraw the same moment.
frame = step(page, TICK_MS if i == 0 else 0, grab=True)
frame.save(out / name / f"{tick:04d}.png")
if name == "y000":
front_masks.append(np.asarray(frame)[:, :, 3] > 127)
if tick % 20 == 0:
print(f"tick {tick}/{ticks}")
period, err = loop_ticks(front_masks)
meta = {
"url": args.url,
"skin": args.skin,
"tick_ms": TICK_MS,
"ticks": ticks,
"start_ms": start,
"size": list(front_masks[0].shape[::-1]),
"aspect": info["aspect"],
"near": info["near"],
"views": cameras,
"loop_ticks": period,
"loop_error": err,
}
(out / "views.json").write_text(json.dumps(meta, indent=2))
print(f"loop: {period} ticks ({period * TICK_MS / 1000:.2f} s), mismatch {err:.4f}")
browser.close()
if __name__ == "__main__":
main()