Files
selimaj-devandclaude df9a2587e6 Add a tool that measures emotes from a store's 3D preview
tools/capture drives a store page's skin viewer in headless Firefox, stepping its clock one tick at
a time and taking every tick from ten fixed cameras. It then fits our own rig to the frames by
rendering the model (a Python port of LimbBend and PoseApplier) and matching outlines and colours,
and writes the result as Emotecraft JSON. It measures pixels only and never reads the page's
animation data. The README covers the steps, the checks and the limits.

Co-Authored-By: Claude Opus 5.5 <[email protected]>
2026-09-29 16:55:08 +02:00

175 lines
7.3 KiB
Python

"""Renders model points through the cameras capture.py used, and scores them against captured masks."""
import json
from pathlib import Path
import numpy as np
from PIL import Image
from scipy.ndimage import distance_transform_edt
# Model space (Y down, facing -Z) to the viewer's world (Y up, facing +Z): Minecraft's scale(-1, -1, 1)
# after turning the player 180 degrees to face south.
FLIP = np.diag([1.0, -1.0, -1.0])
class Camera:
"""A three.js PerspectiveCamera placed with lookAt, as viewer.js places it."""
def __init__(self, view, size, aspect, scale):
yaw, pitch = np.radians(view["yaw"]), np.radians(view["pitch"])
target = np.array(view["target"], float)
d = view["distance"]
self.position = target + d * np.array([np.cos(pitch) * np.sin(yaw), np.sin(pitch), np.cos(pitch) * np.cos(yaw)])
z = self.position - target
z /= np.linalg.norm(z)
x = np.cross([0, 1, 0], z)
x /= np.linalg.norm(x)
y = np.cross(z, x)
self.rotation = np.stack([x, y, z]) # world to camera
self.f = 1 / np.tan(np.radians(view["fov"]) / 2)
self.aspect = aspect
self.width, self.height = size[0] // scale, size[1] // scale
def project(self, world):
c = (world - self.position) @ self.rotation.T
depth = -c[:, 2]
nx = self.f / self.aspect * c[:, 0] / depth
ny = self.f * c[:, 1] / depth
return np.stack([(nx + 1) / 2 * self.width, (1 - ny) / 2 * self.height], axis=1)
def to_world(points, calib):
"""calib = (scale, tx, ty, tz): where the viewer puts the model."""
return calib[0] * points @ FLIP.T + np.asarray(calib[1:4])
def splat(px, width, height):
"""A mask covering every projected point, 2x2 pixels each so the surface has no holes."""
mask = np.zeros((height, width), bool)
x = np.floor(px[:, 0] - 0.5).astype(int)
y = np.floor(px[:, 1] - 0.5).astype(int)
for dx in (0, 1):
for dy in (0, 1):
xs, ys = x + dx, y + dy
ok = (xs >= 0) & (xs < width) & (ys >= 0) & (ys < height)
mask[ys[ok], xs[ok]] = True
return mask
def color_features(rgb):
"""Lighting-tolerant colour: hue and saturation as opponent channels divided by brightness, plus
log brightness. Shading scales all three channels, which leaves the first two alone."""
lum = rgb.mean(axis=-1)
o1 = rgb[..., 0] - rgb[..., 1]
o2 = (rgb[..., 0] + rgb[..., 1]) / 2 - rgb[..., 2]
return np.stack([o1 / (lum + 0.05), o2 / (lum + 0.05), np.log(lum + 0.05)], axis=-1)
class Target:
"""One captured view of one tick, downscaled, with the distance maps the score needs."""
def __init__(self, mask, rgb=None):
self.mask = mask
self.outside = distance_transform_edt(~mask)
self.ys, self.xs = np.nonzero(mask)
self.features = None if rgb is None else color_features(rgb)
def downscale(alpha, scale):
h, w = alpha.shape
blocks = alpha[: h // scale * scale, : w // scale * scale].reshape(h // scale, scale, w // scale, scale)
return blocks.mean(axis=(1, 3)) > 127
def downscale_rgb(rgba, scale):
"""Mean colour of the covered pixels in each block (0-1)."""
h, w = rgba.shape[:2]
blocks = rgba[: h // scale * scale, : w // scale * scale].astype(float)
blocks = blocks.reshape(h // scale, scale, w // scale, scale, 4)
alpha = blocks[..., 3:] / 255
return (blocks[..., :3] / 255 * alpha).sum(axis=(1, 3)) / np.maximum(alpha.sum(axis=(1, 3)), 1e-6)
class Capture:
"""The frames and cameras of one capture."""
def __init__(self, directory, scale=4, views=None):
self.dir = Path(directory)
self.meta = json.loads((self.dir / "views.json").read_text())
self.scale = scale
self.names = views or list(self.meta["views"])
self.cameras = {n: Camera(self.meta["views"][n], self.meta["size"], self.meta["aspect"], scale) for n in self.names}
def targets(self, tick):
out = {}
for n in self.names:
rgba = np.asarray(Image.open(self.dir / n / f"{tick:04d}.png"))
out[n] = Target(downscale(rgba[:, :, 3], self.scale), downscale_rgb(rgba, self.scale))
return out
# How much a unit of colour difference counts against a pixel of outline distance, and how much
# brightness counts next to hue and saturation.
COLOR_WEIGHT = 0.5
BRIGHTNESS_WEIGHT = 0.3
def score(points, calib, cameras, targets, per_view=False, colors=None, color_views=None, one_way=False,
color_weight=None):
"""Mean outline distance in pixels, both ways (our points outside theirs and theirs outside ours),
plus, with colours, how different the colours are where both outlines cover a pixel.
one_way only counts our points outside theirs, for fitting some of the parts: the rest of their
outline is still unexplained, and shouldn't pull the parts being fitted towards it."""
world = to_world(points, calib)
total = {}
for name, cam in cameras.items():
t = targets[name]
px = cam.project(world)
xi = np.clip(px[:, 0].astype(int), 0, cam.width - 1)
yi = np.clip(px[:, 1].astype(int), 0, cam.height - 1)
ours_out = t.outside[yi, xi].mean()
if colors is not None and t.features is not None and (color_views is None or name in color_views):
depth = -((world - cam.position) @ cam.rotation.T)[:, 2]
image, mask = splat_color(px, depth, colors, cam.width, cam.height)
both = mask & t.mask
diff = np.abs(color_features(image[both]) - t.features[both])
color = (diff[:, 0] + diff[:, 1] + BRIGHTNESS_WEIGHT * diff[:, 2]).mean() if both.any() else 0
else:
mask = splat(px, cam.width, cam.height)
color = 0
theirs_out = 0 if one_way or not len(t.ys) else distance_transform_edt(~mask)[t.ys, t.xs].mean()
total[name] = ours_out + theirs_out + (COLOR_WEIGHT if color_weight is None else color_weight) * color
return total if per_view else sum(total.values()) / len(total)
def render_masks(points, calib, cameras):
world = to_world(points, calib)
return {n: splat(c.project(world), c.width, c.height) for n, c in cameras.items()}
def splat_color(px, depth, colors, width, height):
"""Colour image of the nearest point in each pixel (2x2 pixels per point), and its coverage mask."""
order = np.argsort(-depth) # far to near, so nearer points are written last and win
x = np.floor(px[order, 0] - 0.5).astype(int)
y = np.floor(px[order, 1] - 0.5).astype(int)
# All four pixels of every point in one write; a write per offset would let far points overwrite
# pixels near points filled at another offset.
xs = np.stack([x, x + 1, x, x + 1], axis=1).ravel()
ys = np.stack([y, y, y + 1, y + 1], axis=1).ravel()
cs = np.repeat(colors[order], 4, axis=0)
ok = (xs >= 0) & (xs < width) & (ys >= 0) & (ys < height)
image = np.zeros((height, width, 3))
mask = np.zeros((height, width), bool)
image[ys[ok], xs[ok]] = cs[ok]
mask[ys[ok], xs[ok]] = True
return image, mask
def render_colors(points, colors, calib, cameras):
world = to_world(points, calib)
out = {}
for n, c in cameras.items():
depth = -((world - c.position) @ c.rotation.T)[:, 2]
out[n] = splat_color(c.project(world), depth, colors, c.width, c.height)
return out