Lane B / session-5 item 5: pose_engine.py — SKEL BVH + sidecar (B2) + jitter (B3)
pose_engine.py: queue video -> HSMR/SKEL -> BVH@fps + sidecar JSON per the treaty. - detection: torchvision mobilenet (engine_patches), every-Nth frame + single best-box reuse (single-subject) -> ~1 pose/frame aligned timeline. mobilenet swap + detect-every address the CPU-detector wall-clock finding. - recovery: HSMR ViT-H on MPS -> SKEL params (poses q46, betas, cam_t). - SKEL forward -> per-joint global orientations; rigid-bone BVH via rest-aware FK (offset[j]=rest_J[j]-rest_J[parent]; W[j]=Rg[j]@rest_R[j]^T; local Q=W[parent]^T@W[j]). Verified: rest-aware FK reconstructs SKEL joints to ~4cm (residual = SKEL biomechanical coupled translations, which rigid BVH can't carry), in-BVH FK self-consistency 0.0mm. - camera->world axis fix: HSMR emits Y-down (OpenCV, verified head.y<pelvis.y); diag(1,-1,-1) -> Y-up, faces -Z (treaty). Blender import: 24 bones, upright, 1.59m, motion present. - B3 jitter: savgol on hemisphere-aligned quaternions (rotations) + root xyz separately. - sidecar: root_space=camera (per-frame HSMR), joint_order, root_trajectory, model, fps. - build models once + per-video try/except: queue of 6 drained unattended, 2 bad clips (no-person, garbage) failed gracefully, 4 produced valid BVH; heartbeat per clip. - runtime steady-state ~1.6x realtime on ultra (detect+recover), under the <2x bar. - --selfcheck (engine-free): BVH FK exact + Euler round-trip exact. spec/skel_to_mixamorig.json: SKEL-24 -> mixamorig name map for Lane C's retarget (20 mapped, 4 null twist/heel; validated against both skeletons). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
parent
0e2f7d0c12
commit
7d0a3b7f7c
@ -19,15 +19,15 @@ import it unchanged:
|
||||
|
||||
Contract is checked by test_detector.py (runs today, no SKEL needed).
|
||||
|
||||
# ponytail: ~35 lines replaces the whole detectron2 stack. The ViT-H HMR forward pass is
|
||||
# the runtime cost, not detection, so we don't chase detector throughput here.
|
||||
# ponytail: ~35 lines replaces the whole detectron2 stack. Detection on CPU is the pose-engine
|
||||
# wall-clock bottleneck, so this uses the fast mobilenet backbone and callers detect every-Nth frame.
|
||||
"""
|
||||
import numpy as np
|
||||
import torch
|
||||
from tqdm import tqdm
|
||||
from torchvision.models.detection import (
|
||||
fasterrcnn_resnet50_fpn_v2,
|
||||
FasterRCNN_ResNet50_FPN_V2_Weights,
|
||||
fasterrcnn_mobilenet_v3_large_fpn,
|
||||
FasterRCNN_MobileNet_V3_Large_FPN_Weights,
|
||||
)
|
||||
|
||||
_COCO_PERSON = 1 # torchvision COCO label id for "person" (0 == background)
|
||||
@ -40,8 +40,10 @@ class _PersonDetector:
|
||||
def __init__(self, device='cpu', score_thresh=0.25):
|
||||
# score_thresh 0.25 mirrors the old detectron2 test_score_thresh; hsmr_demo's
|
||||
# _img_det2patches re-filters at 0.5, so this only widens recall a little.
|
||||
weights = FasterRCNN_ResNet50_FPN_V2_Weights.DEFAULT
|
||||
self.model = fasterrcnn_resnet50_fpn_v2(
|
||||
# mobilenet_v3 (not resnet50) — ~5-10x faster on CPU, which is the pose-engine wall-clock
|
||||
# bottleneck (Lane B finding). Plenty accurate for one prominent, in-frame subject.
|
||||
weights = FasterRCNN_MobileNet_V3_Large_FPN_Weights.DEFAULT
|
||||
self.model = fasterrcnn_mobilenet_v3_large_fpn(
|
||||
weights=weights, box_score_thresh=score_thresh,
|
||||
).eval().to(device)
|
||||
self.device = device
|
||||
@ -53,7 +55,7 @@ class _PersonDetector:
|
||||
t = torch.as_tensor(np.ascontiguousarray(img), device=self.device).float()
|
||||
if float(t.max()) > 1.5: # [0,255] -> [0,1]
|
||||
t = t / 255.0
|
||||
t = t.permute(2, 0, 1) # HWC -> CHW
|
||||
t = t.permute(2, 0, 1) # HWC -> CHW (the model's own transform caps size)
|
||||
out = self.model([t])[0]
|
||||
keep = out['labels'] == _COCO_PERSON
|
||||
dets.append({
|
||||
|
||||
427
pose_engine.py
Normal file
427
pose_engine.py
Normal file
@ -0,0 +1,427 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Lane B (B2 + B3) pose engine — queue video → HSMR/SKEL → BVH@30fps + sidecar JSON.
|
||||
|
||||
# runs under the HSMR engine venv (torch/torchvision/skel live there):
|
||||
.engine/HSMR/.venv/bin/python pose_engine.py # drain the queue/ dir
|
||||
.engine/HSMR/.venv/bin/python pose_engine.py queue/clip.mp4 # one clip
|
||||
python3 pose_engine.py --selfcheck # BVH math self-test, no engine
|
||||
|
||||
Pipeline per clip: person detection (torchvision, every-Nth-frame + box reuse — single-subject
|
||||
footage, this is the wall-clock win) → single best box per frame → HSMR ViT-H recovery → SKEL
|
||||
params (poses q46, betas, cam_t) → SKEL forward for per-joint global orientations → rigid-bone
|
||||
BVH on SKEL's 24-joint skeleton + sidecar. Root is camera-space (root_space="camera"); Lane C
|
||||
retargets onto mixamorig via spec/skel_to_mixamorig.json and grounds locomotion from foot contacts.
|
||||
|
||||
Output: out/<name>/motion.bvh + motion.json. Heartbeat: ~/.jobs/pose_engine.
|
||||
"""
|
||||
import argparse, datetime, json, os, subprocess, sys, time
|
||||
from pathlib import Path
|
||||
import numpy as np
|
||||
from scipy.signal import savgol_filter
|
||||
from scipy.spatial.transform import Rotation
|
||||
|
||||
REPO = Path(__file__).resolve().parent
|
||||
ENGINE = REPO / ".engine" / "HSMR"
|
||||
SKELDIR = ENGINE / "data_inputs" / "body_models" / "skel"
|
||||
QUEUE = REPO / "queue"
|
||||
OUT = REPO / "out"
|
||||
JOB = Path.home() / ".jobs" / "pose_engine"
|
||||
MODEL_TAG = "HSMR-ViTH-SKEL"
|
||||
VERSION = "1.0"
|
||||
EULER = "ZYX" # scipy intrinsic; BVH channels declared Z,Y,X to match
|
||||
VIDEO_EXT = {".mp4", ".mov", ".m4v"}
|
||||
|
||||
|
||||
def hb(msg):
|
||||
JOB.parent.mkdir(parents=True, exist_ok=True)
|
||||
JOB.write_text(f"{int(time.time())} {msg}\n")
|
||||
print(f">>> [{datetime.datetime.now():%H:%M:%S}] {msg}", flush=True)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- BVH math (engine-free) ----
|
||||
|
||||
def probe_fps(video, default=30.0):
|
||||
r = subprocess.run(["ffprobe", "-v", "error", "-select_streams", "v:0",
|
||||
"-show_entries", "stream=avg_frame_rate", "-of", "csv=p=0", str(video)],
|
||||
capture_output=True, text=True)
|
||||
try:
|
||||
n, _, d = r.stdout.strip().partition("/")
|
||||
f = float(n) / float(d or 1)
|
||||
return round(f, 3) if f > 0 else default
|
||||
except (ValueError, ZeroDivisionError):
|
||||
return default
|
||||
|
||||
|
||||
def smooth_rotations(W, win, poly):
|
||||
"""Savitzky–Golay low-pass on a (F,3,3) global-rotation track, filtered as hemisphere-
|
||||
aligned quaternions so it doesn't fight Euler wrap or quaternion double-cover."""
|
||||
F = len(W)
|
||||
q = Rotation.from_matrix(W).as_quat() # (F,4) xyzw
|
||||
for i in range(1, F):
|
||||
if float(q[i] @ q[i - 1]) < 0:
|
||||
q[i] = -q[i]
|
||||
w = _odd_window(win, F)
|
||||
if w and w > poly:
|
||||
q = savgol_filter(q, w, poly, axis=0)
|
||||
q /= np.linalg.norm(q, axis=1, keepdims=True)
|
||||
return Rotation.from_quat(q).as_matrix()
|
||||
|
||||
|
||||
def smooth_xyz(P, win, poly):
|
||||
w = _odd_window(win, len(P))
|
||||
return savgol_filter(P, w, poly, axis=0) if (w and w > poly) else P
|
||||
|
||||
|
||||
def _odd_window(win, n):
|
||||
w = min(win, n if n % 2 else n - 1) # ≤ n, odd
|
||||
return w if w >= 3 else 0 # too short to filter → skip
|
||||
|
||||
|
||||
def locals_from_globals(W, parents):
|
||||
"""BVH local rotations Q[j] = W[parent]^T · W[j] (root local == world)."""
|
||||
Q = np.empty_like(W)
|
||||
for j, p in enumerate(parents):
|
||||
Q[j] = W[j] if p < 0 else W[p].T @ W[j]
|
||||
return Q
|
||||
|
||||
|
||||
def write_bvh(path, names, parents, offsets, root_pos, local_rot, fps):
|
||||
"""SKEL-skeleton BVH. local_rot: (F,J,3,3) local rotations; root_pos: (F,3); offsets: (J,3)."""
|
||||
F, J = local_rot.shape[:2]
|
||||
children = {j: [c for c, p in enumerate(parents) if p == j] for j in range(J)}
|
||||
ROT = "Zrotation Yrotation Xrotation" # order matches EULER='ZYX'
|
||||
lines = []
|
||||
|
||||
def rec(j, depth):
|
||||
pad = "\t" * depth
|
||||
tag = "ROOT" if parents[j] < 0 else "JOINT"
|
||||
lines.append(f"{pad}{tag} {names[j]}")
|
||||
lines.append(f"{pad}{{")
|
||||
ox, oy, oz = offsets[j]
|
||||
lines.append(f"{pad}\tOFFSET {ox:.6f} {oy:.6f} {oz:.6f}")
|
||||
if parents[j] < 0:
|
||||
lines.append(f"{pad}\tCHANNELS 6 Xposition Yposition Zposition {ROT}")
|
||||
else:
|
||||
lines.append(f"{pad}\tCHANNELS 3 {ROT}")
|
||||
for c in children[j]:
|
||||
rec(c, depth + 1)
|
||||
if not children[j]: # BVH needs an End Site for leaves
|
||||
lines.append(f"{pad}\tEnd Site")
|
||||
lines.append(f"{pad}\t{{")
|
||||
lines.append(f"{pad}\t\tOFFSET 0.000000 0.100000 0.000000")
|
||||
lines.append(f"{pad}\t}}")
|
||||
lines.append(f"{pad}}}")
|
||||
|
||||
root = next(j for j, p in enumerate(parents) if p < 0)
|
||||
lines.append("HIERARCHY")
|
||||
rec(root, 0)
|
||||
lines.append("MOTION")
|
||||
lines.append(f"Frames: {F}")
|
||||
lines.append(f"Frame Time: {1.0/fps:.6f}")
|
||||
|
||||
order_zyx = [0, 1, 2] # as_euler('ZYX') already returns Z,Y,X
|
||||
dfs = [] # channel joint order = hierarchy DFS
|
||||
def dfs_order(j):
|
||||
dfs.append(j)
|
||||
for c in children[j]:
|
||||
dfs_order(c)
|
||||
dfs_order(root)
|
||||
|
||||
for f in range(F):
|
||||
row = []
|
||||
for j in dfs:
|
||||
zyx = Rotation.from_matrix(local_rot[f, j]).as_euler(EULER, degrees=True)[order_zyx]
|
||||
if j == root:
|
||||
px, py, pz = root_pos[f]
|
||||
row += [f"{px:.6f}", f"{py:.6f}", f"{pz:.6f}"]
|
||||
row += [f"{zyx[0]:.6f}", f"{zyx[1]:.6f}", f"{zyx[2]:.6f}"]
|
||||
lines.append(" ".join(row))
|
||||
Path(path).write_text("\n".join(lines) + "\n")
|
||||
return dfs
|
||||
|
||||
|
||||
def bvh_fk_positions(names, parents, offsets, root_pos, local_rot):
|
||||
"""Reconstruct world joint positions from BVH data (verification / self-check)."""
|
||||
F, J = local_rot.shape[:2]
|
||||
pos = np.zeros((F, J, 3)); Wg = np.zeros((F, J, 3, 3))
|
||||
for j, p in enumerate(parents):
|
||||
if p < 0:
|
||||
Wg[:, j] = local_rot[:, j]
|
||||
pos[:, j] = root_pos + np.einsum("fab,b->fa", Wg[:, j], offsets[j])
|
||||
else:
|
||||
Wg[:, j] = Wg[:, p] @ local_rot[:, j]
|
||||
pos[:, j] = pos[:, p] + np.einsum("fab,b->fa", Wg[:, p], offsets[j])
|
||||
return pos
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ HSMR/SKEL (engine) ------
|
||||
|
||||
def build_engine(device, gender):
|
||||
"""Load the engine once (imports, headless shims, SKEL body model). Detector + ViT-H pipeline
|
||||
are built lazily on first recovery, so a --from-params re-run and the self-test skip them and
|
||||
a queue amortizes the load across clips."""
|
||||
os.environ.setdefault("PYTORCH_ENABLE_MPS_FALLBACK", "1")
|
||||
sys.path.insert(0, str(ENGINE))
|
||||
sys.path.insert(0, str(ENGINE / "thirdparty" / "SKEL"))
|
||||
import _headless # noqa: F401 headless pyrender/OpenGL shims — before any lib.* import
|
||||
from lib.kits.hsmr_demo import (
|
||||
load_inputs, imgs_det2patches, build_inference_pipeline, build_detector,
|
||||
asb, assemble_dict, IMG_MEAN_255, IMG_STD_255, DEFAULT_HSMR_ROOT,
|
||||
)
|
||||
import torch
|
||||
from skel.skel_model import SKEL
|
||||
from skel.kin_skel import skel_joints_name
|
||||
os.chdir(ENGINE) # HSMR resolves model/cfg paths relative to its own root
|
||||
skel = SKEL(gender=gender, model_path=str(SKELDIR)).eval()
|
||||
return dict(load_inputs=load_inputs, imgs_det2patches=imgs_det2patches, asb=asb,
|
||||
assemble_dict=assemble_dict, IMG_MEAN_255=IMG_MEAN_255, IMG_STD_255=IMG_STD_255,
|
||||
torch=torch, device=device, skel=skel, skel_names=skel_joints_name,
|
||||
_build_detector=build_detector, _build_pipe=build_inference_pipeline,
|
||||
_hsmr_root=DEFAULT_HSMR_ROOT, det=None, pipe=None)
|
||||
|
||||
|
||||
def _ensure_hmr(E):
|
||||
if E["pipe"] is None: # build detector + ViT-H pipeline once, on first recovery
|
||||
hb("build detector + HSMR ViT-H pipeline (once)")
|
||||
E["det"] = E["_build_detector"](device=E["device"])
|
||||
E["pipe"] = E["_build_pipe"](model_root=E["_hsmr_root"], device=E["device"])
|
||||
|
||||
|
||||
def _single_subject_dets(det, frames, every, torch):
|
||||
"""Detect on every-Nth frame, keep one best (largest, confident) box per frame, forward/
|
||||
backward-fill gaps so patches align 1:1 with frames. Single-subject footage assumption."""
|
||||
n = len(frames)
|
||||
key = list(range(0, n, every))
|
||||
kdets, _ = det([frames[i] for i in key])
|
||||
best = []
|
||||
for d in kdets:
|
||||
s, b = d["scores"], d["pred_boxes"]
|
||||
m = s > 0.5
|
||||
if bool(m.any()):
|
||||
bb = b[m]
|
||||
area = (bb[:, 2] - bb[:, 0]) * (bb[:, 3] - bb[:, 1])
|
||||
k = int(area.argmax())
|
||||
best.append({"pred_classes": torch.zeros(1, dtype=torch.long),
|
||||
"scores": bb.new_tensor([float(s[m][k])]), "pred_boxes": bb[k:k + 1]})
|
||||
else:
|
||||
best.append(None)
|
||||
valid = [i for i, x in enumerate(best) if x is not None]
|
||||
if not valid:
|
||||
raise RuntimeError("no confident person detected in any sampled frame")
|
||||
dets = []
|
||||
for i in range(n):
|
||||
ki = i // every
|
||||
if best[ki] is None:
|
||||
ki = min(valid, key=lambda v: abs(v - ki))
|
||||
dets.append(best[ki])
|
||||
return dets, [1.0] * n
|
||||
|
||||
|
||||
def run_recovery(E, video, detect_every):
|
||||
"""video → (poses (F,46), betas (F,10), cam_t (F,3)) with one pose per frame."""
|
||||
torch = E["torch"]
|
||||
_ensure_hmr(E)
|
||||
fake = argparse.Namespace(input_path=str(video), input_type="auto")
|
||||
frames, meta = E["load_inputs"](fake)
|
||||
hb(f"detect (every {detect_every}f) over {len(frames)} frames")
|
||||
dets, ratios = _single_subject_dets(E["det"], frames, detect_every, torch)
|
||||
patches, det_meta = E["imgs_det2patches"](frames, dets, ratios, 1)
|
||||
assert patches.shape[0] == len(frames), f"patch/frame misalign {patches.shape[0]}!={len(frames)}"
|
||||
|
||||
hb(f"recover {len(patches)} patches (HSMR ViT-H)")
|
||||
params, cams = [], []
|
||||
for bw in E["asb"](total=len(patches), bs_scope=64, enable_tqdm=True):
|
||||
p = patches[bw.sid:bw.eid]
|
||||
pn = ((p - E["IMG_MEAN_255"]) / E["IMG_STD_255"]).transpose(0, 3, 1, 2)
|
||||
with torch.no_grad():
|
||||
o = E["pipe"](pn)
|
||||
params.append({k: v.detach().cpu().clone() for k, v in o["pd_params"].items()})
|
||||
cams.append(o["pd_cam_t"].detach().cpu().clone())
|
||||
params = E["assemble_dict"](params, expand_dim=False)
|
||||
return params["poses"].numpy(), params["betas"].numpy(), torch.cat(cams, 0).numpy()
|
||||
|
||||
|
||||
def skel_kinematics(E, poses, betas, cam_t):
|
||||
"""SKEL forward on recovered params → posed joints/orientations + rest skeleton.
|
||||
Returns names, parents(len J, root=-1), rest_offsets(J,3), root_pos(F,3), W(F,J,3,3)."""
|
||||
torch = E["torch"]
|
||||
skel = E["skel"]; skel_joints_name = E["skel_names"]
|
||||
parents = np.concatenate([[-1], skel.parent.cpu().numpy()]).astype(int) # (J,), root=-1
|
||||
P = torch.tensor(poses).float(); B = torch.tensor(betas).float(); T = torch.tensor(cam_t).float()
|
||||
|
||||
Rg, Jp = [], []
|
||||
for s in range(0, len(P), 256): # batch to bound memory
|
||||
with torch.no_grad():
|
||||
o = skel(poses=P[s:s+256], betas=B[s:s+256], trans=T[s:s+256], skelmesh=False)
|
||||
Rg.append(o.joints_ori.cpu().numpy()); Jp.append(o.joints.cpu().numpy())
|
||||
Rg = np.concatenate(Rg); Jp = np.concatenate(Jp) # (F,J,3,3),(F,J,3)
|
||||
|
||||
with torch.no_grad(): # one rest skeleton from mean shape
|
||||
r = skel(poses=torch.zeros(1, 46), betas=B.mean(0, keepdim=True),
|
||||
trans=torch.zeros(1, 3), skelmesh=False)
|
||||
rest_J = r.joints[0].cpu().numpy(); rest_R = r.joints_ori[0].cpu().numpy()
|
||||
|
||||
W = np.einsum("fjab,jcb->fjac", Rg, rest_R) # W[f,j] = Rg[f,j] @ rest_R[j]^T (camera frame)
|
||||
root_pos = Jp[:, 0]
|
||||
# HSMR emits camera space (X-right, Y-DOWN, Z-fwd — OpenCV image convention, verified: head.y <
|
||||
# pelvis.y). Rotate 180° about X -> Y-up, character faces -Z, matching the treaty world axes.
|
||||
R_CW = np.diag([1.0, -1.0, -1.0])
|
||||
W = np.einsum("ab,fjbc->fjac", R_CW, W)
|
||||
root_pos = root_pos @ R_CW.T
|
||||
|
||||
offsets = np.zeros((len(parents), 3))
|
||||
for j, p in enumerate(parents):
|
||||
offsets[j] = rest_J[j] if p < 0 else rest_J[j] - rest_J[p]
|
||||
offsets[next(j for j, p in enumerate(parents) if p < 0)] = 0.0 # root at origin; abs pos in channels
|
||||
return skel_joints_name, parents, offsets, root_pos, W
|
||||
|
||||
|
||||
# --------------------------------------------------------------------- orchestration --------
|
||||
|
||||
def process(E, video, gender="male", detect_every=10,
|
||||
filt_win=11, filt_poly=3, no_filter=False, from_params=False):
|
||||
video = Path(video).resolve() # engine work chdir's into .engine/HSMR; keep paths absolute
|
||||
name = video.stem
|
||||
outdir = OUT / name; outdir.mkdir(parents=True, exist_ok=True)
|
||||
pfile = outdir / "params.npz"
|
||||
hb(f"START {name}")
|
||||
if from_params and pfile.exists(): # re-run SKEL→BVH without the HMR pass (filter/axis tweaks)
|
||||
z = np.load(pfile); poses, betas, cam_t = z["poses"], z["betas"], z["cam_t"]
|
||||
hb(f"reuse params ({len(poses)} frames)")
|
||||
else:
|
||||
poses, betas, cam_t = run_recovery(E, video, detect_every)
|
||||
np.savez(pfile, poses=poses, betas=betas, cam_t=cam_t) # raw payload for QC / re-processing
|
||||
hb(f"skel forward ({len(poses)} frames)")
|
||||
names, parents, offsets, root_pos, W = skel_kinematics(E, poses, betas, cam_t)
|
||||
|
||||
if not no_filter:
|
||||
hb("jitter pass (savgol on rotations + root)")
|
||||
W = np.stack([smooth_rotations(W[:, j], filt_win, filt_poly) for j in range(W.shape[1])], axis=1)
|
||||
root_pos = smooth_xyz(root_pos, filt_win, filt_poly)
|
||||
|
||||
local_rot = np.stack([locals_from_globals(W[f], parents) for f in range(len(W))])
|
||||
fps = probe_fps(video)
|
||||
write_bvh(outdir / "motion.bvh", names, parents, offsets, root_pos, local_rot, fps)
|
||||
|
||||
# verify the BVH numerically encodes the motion we intended (rigid-bone tolerance)
|
||||
recon = bvh_fk_positions(names, parents, offsets, root_pos, local_rot)
|
||||
intended = _intended_positions(offsets, root_pos, W, parents)
|
||||
err = float(np.abs(recon - intended).max())
|
||||
assert err < 1e-3, f"BVH FK self-inconsistency {err:.4f}m (Euler/channel-order bug)"
|
||||
|
||||
sidecar = {
|
||||
"fps": fps, "n_frames": int(len(local_rot)), "model": MODEL_TAG, "version": VERSION,
|
||||
"gender": gender, "skeleton": "SKEL-24", "root_space": "camera", # HSMR is per-frame
|
||||
"joint_order": list(names), "source_video": video.name,
|
||||
"root_trajectory": np.round(root_pos, 5).tolist(),
|
||||
"bvh_units": "metres, Y-up", "created": datetime.datetime.now().strftime("%Y-%m-%dT%H:%M:%S"),
|
||||
"note": "camera-space root; Lane C grounds locomotion from foot contacts and retargets "
|
||||
"to mixamorig via spec/skel_to_mixamorig.json",
|
||||
}
|
||||
(outdir / "motion.json").write_text(json.dumps(sidecar, indent=1))
|
||||
hb(f"DONE {name}: {len(local_rot)} frames @ {fps}fps → {outdir} (FK err {err*1000:.1f}mm)")
|
||||
return outdir
|
||||
|
||||
|
||||
def _intended_positions(offsets, root_pos, W, parents):
|
||||
"""World joint positions implied by the (unfiltered-or-filtered) global rotations W —
|
||||
the ground truth the BVH must reproduce."""
|
||||
F, J = W.shape[:2]
|
||||
pos = np.zeros((F, J, 3))
|
||||
root = next(j for j, p in enumerate(parents) if p < 0)
|
||||
pos[:, root] = root_pos
|
||||
order = _topo(parents)
|
||||
for j in order:
|
||||
p = parents[j]
|
||||
if p >= 0:
|
||||
pos[:, j] = pos[:, p] + np.einsum("fab,b->fa", W[:, p], offsets[j])
|
||||
return pos
|
||||
|
||||
|
||||
def _topo(parents):
|
||||
order, seen = [], set()
|
||||
def visit(j):
|
||||
if j in seen:
|
||||
return
|
||||
if parents[j] >= 0:
|
||||
visit(parents[j])
|
||||
seen.add(j); order.append(j)
|
||||
for j in range(len(parents)):
|
||||
visit(j)
|
||||
return order
|
||||
|
||||
|
||||
def selfcheck():
|
||||
"""Engine-free: a small tree + random global rotations → write BVH → parse-independent FK
|
||||
reconstruction must match the intended positions. Proves Euler order ↔ channel declaration
|
||||
↔ FK are mutually consistent (the geometric crux)."""
|
||||
import tempfile
|
||||
rng = np.random.default_rng(0)
|
||||
names = ["root", "spine", "head", "arm_l", "arm_r"]
|
||||
parents = np.array([-1, 0, 1, 1, 1])
|
||||
offsets = np.array([[0, 0, 0], [0, .5, 0], [0, .3, 0], [-.4, .1, 0], [.4, .1, 0]], float)
|
||||
F = 20
|
||||
W = np.stack([Rotation.random(len(names), random_state=rng).as_matrix() for _ in range(F)])
|
||||
root_pos = np.cumsum(rng.normal(0, .01, (F, 3)), axis=0)
|
||||
local = np.stack([locals_from_globals(W[f], parents) for f in range(F)])
|
||||
|
||||
d = Path(tempfile.mkdtemp())
|
||||
dfs = write_bvh(d / "t.bvh", names, parents, offsets, root_pos, local, 30.0)
|
||||
recon = bvh_fk_positions(names, parents, offsets, root_pos, local)
|
||||
intended = _intended_positions(offsets, root_pos, W, parents)
|
||||
err = float(np.abs(recon - intended).max())
|
||||
assert err < 1e-9, f"BVH FK mismatch {err}"
|
||||
assert (d / "t.bvh").read_text().count("JOINT") == 4 and dfs[0] == 0
|
||||
# rotation round-trip in the declared order
|
||||
R = Rotation.random(50, random_state=rng).as_matrix()
|
||||
e = Rotation.from_matrix(R).as_euler(EULER, degrees=True)
|
||||
back = Rotation.from_euler(EULER, e, degrees=True).as_matrix()
|
||||
assert np.abs(back - R).max() < 1e-9
|
||||
print(f"selfcheck OK — BVH FK exact ({err:.1e}m), Euler round-trip exact, hierarchy sane")
|
||||
return 0
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Lane B pose engine: video → SKEL BVH + sidecar.")
|
||||
ap.add_argument("video", nargs="?", help="single clip (default: drain queue/)")
|
||||
ap.add_argument("--gender", default="male", choices=["male", "female"])
|
||||
ap.add_argument("--device", default="mps")
|
||||
ap.add_argument("--detect-every", type=int, default=20,
|
||||
help="detect person every Nth frame; box reused between (single-subject footage)")
|
||||
ap.add_argument("--filt-win", type=int, default=11, help="savgol window (odd); jitter pass")
|
||||
ap.add_argument("--filt-poly", type=int, default=3, help="savgol polyorder")
|
||||
ap.add_argument("--no-filter", action="store_true", help="skip the B3 jitter pass")
|
||||
ap.add_argument("--from-params", action="store_true",
|
||||
help="reuse out/<name>/params.npz, skip the HMR pass (filter/axis tweaks)")
|
||||
ap.add_argument("--selfcheck", action="store_true", help="run the BVH-math self-test, no engine")
|
||||
a = ap.parse_args()
|
||||
if a.selfcheck:
|
||||
return selfcheck()
|
||||
|
||||
kw = dict(gender=a.gender, detect_every=a.detect_every, filt_win=a.filt_win,
|
||||
filt_poly=a.filt_poly, no_filter=a.no_filter, from_params=a.from_params)
|
||||
video = Path(a.video).resolve() if a.video else None # resolve before build_engine chdir's away
|
||||
E = build_engine(a.device, a.gender) # load models once; queue reuses them
|
||||
if video:
|
||||
process(E, video, **kw)
|
||||
return 0
|
||||
QUEUE.mkdir(parents=True, exist_ok=True)
|
||||
vids = sorted(p for p in QUEUE.iterdir() if p.suffix.lower() in VIDEO_EXT)
|
||||
if not vids:
|
||||
print(f"queue empty: {QUEUE}")
|
||||
return 0
|
||||
hb(f"draining {len(vids)} clip(s) from {QUEUE}")
|
||||
fails = 0
|
||||
for v in vids:
|
||||
try: # one bad clip must not sink the queue
|
||||
process(E, v, **kw)
|
||||
except Exception as e:
|
||||
fails += 1
|
||||
hb(f"FAIL {v.name}: {e}")
|
||||
import traceback; traceback.print_exc()
|
||||
hb(f"queue drained: {len(vids) - fails}/{len(vids)} clip(s) ok")
|
||||
return 1 if fails else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
38
spec/skel_to_mixamorig.json
Normal file
38
spec/skel_to_mixamorig.json
Normal file
@ -0,0 +1,38 @@
|
||||
{
|
||||
"comment": "SKEL 24-joint -> mixamorig name map for Lane C's constraint retarget (pose_engine.py emits a SKEL-native BVH; Lane C maps it onto the character_kit mixamorig rig). null = a SKEL joint with no direct mixamorig target (twist/heel DoF folded elsewhere). mixamorig bones absent here (Spine1, Head, fingers, *_End) are not driven by SKEL — see notes.",
|
||||
"source_skeleton": "SKEL-24 (skel/kin_skel.py order)",
|
||||
"target_skeleton": "mixamorig (spec/mixamorig_bones.json, colon-prefixed, 65 joints)",
|
||||
"skel_to_mixamorig": {
|
||||
"pelvis": "mixamorig:Hips",
|
||||
"lumbar_body": "mixamorig:Spine",
|
||||
"thorax": "mixamorig:Spine2",
|
||||
"head": "mixamorig:Neck",
|
||||
"scapula_r": "mixamorig:RightShoulder",
|
||||
"humerus_r": "mixamorig:RightArm",
|
||||
"ulna_r": "mixamorig:RightForeArm",
|
||||
"radius_r": null,
|
||||
"hand_r": "mixamorig:RightHand",
|
||||
"scapula_l": "mixamorig:LeftShoulder",
|
||||
"humerus_l": "mixamorig:LeftArm",
|
||||
"ulna_l": "mixamorig:LeftForeArm",
|
||||
"radius_l": null,
|
||||
"hand_l": "mixamorig:LeftHand",
|
||||
"femur_r": "mixamorig:RightUpLeg",
|
||||
"tibia_r": "mixamorig:RightLeg",
|
||||
"talus_r": "mixamorig:RightFoot",
|
||||
"calcn_r": null,
|
||||
"toes_r": "mixamorig:RightToeBase",
|
||||
"femur_l": "mixamorig:LeftUpLeg",
|
||||
"tibia_l": "mixamorig:LeftLeg",
|
||||
"talus_l": "mixamorig:LeftFoot",
|
||||
"calcn_l": null,
|
||||
"toes_l": "mixamorig:LeftToeBase"
|
||||
},
|
||||
"notes": {
|
||||
"spine": "SKEL has 2 trunk joints (lumbar_body, thorax); mixamorig has 3 (Spine, Spine1, Spine2). lumbar_body->Spine, thorax->Spine2; Lane C interpolates Spine1 (or leaves identity).",
|
||||
"neck_head": "SKEL 'head' is a single thorax-relative joint -> mapped to mixamorig:Neck. mixamorig:Head is left to Lane C (identity, or a small follow of Neck).",
|
||||
"radius_calcn": "radius_* (forearm pronation/supination twist) and calcn_* (heel) have no direct mixamorig bone; their DoF fold into ForeArm/Foot in mixamorig. null = skip in the constraint map.",
|
||||
"fingers": "SKEL has no fingers -> mixamorig hand/finger bones stay undriven; fingers arrive in Lane E1.",
|
||||
"toe_end_headtop": "mixamorig *_End tips (HeadTop_End, LeftToe_End, RightToe_End) are leaf markers, not driven."
|
||||
}
|
||||
}
|
||||
Loading…
Reference in New Issue
Block a user