component-studio-reference / scripts /retarget_character.py
mantrakp's picture
Isolate reference inference in a dedicated ZeroGPU worker
5c331a4 verified
Raw History Blame Contribute Delete
13.5 kB
"""Retarget generated SOMA directions to a learned Mixamo skin, preserving proportions."""
import argparse
import json
from pathlib import Path
import re
import sys
import bpy
import numpy as np
from mathutils import Matrix, Vector
p = argparse.ArgumentParser()
for name in ["input", "motion", "bvh", "output"]:
p.add_argument("--" + name, required=True)
p.add_argument("--body-clearance", action="store_true")
p.add_argument("--smooth-joints", action="store_true")
p.add_argument("--arm-blend-band", type=float, default=0.55)
a = p.parse_args(sys.argv[sys.argv.index("--") + 1 :])
out = Path(a.output)
out.mkdir(parents=True, exist_ok=True)
bpy.ops.wm.read_factory_settings(use_empty=True)
scene = bpy.context.scene
scene.render.fps = 30
bpy.ops.import_scene.gltf(filepath=str(Path(a.input).resolve()))
rigs = [o for o in scene.objects if o.type == "ARMATURE"]
if len(rigs) != 1:
raise ValueError("Learned rig must contain exactly one skeleton.")
rig = rigs[0]
meshes = [
o
for o in scene.objects
if o.type == "MESH" and any(m.type == "ARMATURE" and m.object == rig for m in o.modifiers)
]
rig.animation_data_clear()
for bone in rig.pose.bones:
bone.matrix_basis.identity()
bpy.context.view_layer.update()
for action in list(bpy.data.actions):
bpy.data.actions.remove(action)
# Normalize the whole hierarchy together, retaining bind matrices and textured geometry.
vertices = np.array([tuple(o.matrix_world @ v.co) for o in meshes for v in o.data.vertices])
low, high = vertices.min(0), vertices.max(0)
height = high[2] - low[2]
if height < 1e-6:
raise ValueError("Degenerate learned rig bounds.")
transform = Matrix.Scale(1 / height, 4) @ Matrix.Translation(
Vector((-(low[0] + high[0]) / 2, -(low[1] + high[1]) / 2, -low[2]))
)
for obj in list(scene.objects):
if obj.parent is None:
obj.matrix_world = transform @ obj.matrix_world
bpy.context.view_layer.update()
# Standardize exported bone names so the geometric auditor can identify extremities.
rename = {"Spine": "Spine1", "Spine1": "Spine2", "Spine2": "Chest", "Neck": "Neck1"}
for side in ["Left", "Right"]:
rename[side + "UpLeg"] = side + "Leg"
rename[side + "Leg"] = side + "Shin"
original = [(b, b.name, b.name.split(":")[-1]) for b in rig.data.bones]
groups_by_bone = {old: [obj.vertex_groups.get(old) for obj in meshes] for bone, old, short in original}
for bone, old, short in original:
staged = "__retarget_" + short
bone.name = staged
for group in groups_by_bone[old]:
if group is not None and group.name != staged:
group.name = staged
for bone, old, short in original:
new = rename.get(short, short)
bone.name = new
for group in groups_by_bone[old]:
if group is not None and group.name != new:
group.name = new
for obj in meshes:
if any(group.name not in rig.data.bones for group in obj.vertex_groups):
raise ValueError("Skin groups do not match retargeted skeleton bone names.")
source_names = re.findall(r"JOINT (\S+)", Path(a.bvh).read_text())
idx = {name: i for i, name in enumerate(source_names)}
with np.load(a.motion, allow_pickle=False) as data:
pos = data["posed_joints"].copy()
source_rotations = data["global_rot_mats"].copy()
contacts = data["foot_contacts"].reshape(len(pos), 2, -1).max(axis=2) > 0.5
C = np.array([[1, 0, 0], [0, 0, -1], [0, 1, 0]])
pos = pos @ C.T
source_rotations = C @ source_rotations @ C.T
if pos.shape[1] != len(source_names):
raise ValueError("Motion NPZ and BVH skeleton disagree.")
tips = {
"Hips": "Spine1",
"Spine1": "Spine2",
"Spine2": "Chest",
"Chest": "Neck1",
"Neck1": "Head",
"Head": "HeadEnd",
}
for side in ["Left", "Right"]:
for bone, tip in [
("Shoulder", "Arm"),
("Arm", "ForeArm"),
("ForeArm", "Hand"),
("Hand", "HandMiddleEnd"),
("Leg", "Shin"),
("Shin", "Foot"),
("Foot", "ToeBase"),
]:
tips[side + bone] = side + tip
required = {"Hips", "Head", "LeftHand", "RightHand", "LeftFoot", "RightFoot"}
if not required.issubset(rig.data.bones.keys()):
raise ValueError("Learned skeleton is not a supported Mixamo humanoid.")
bones = list(rig.data.bones)
heads = {b.name: np.array(b.head_local) for b in bones}
rest = {b.name: b.matrix_local.to_3x3() for b in bones}
torso_points = []
for obj in meshes:
lookup = {group.index: group.name for group in obj.vertex_groups}
matrix = rig.matrix_world.inverted() @ obj.matrix_world
for vertex in obj.data.vertices:
torso_weight = sum(
group.weight
for group in vertex.groups
if lookup[group.group] in {"Hips", "Spine1", "Spine2", "Chest", "LeftLeg", "RightLeg"}
)
if torso_weight > 0.7:
torso_points.append(tuple(matrix @ vertex.co))
if not torso_points:
raise ValueError("Learned skin has no torso region.")
torso_points = np.array(torso_points)
body_low, body_high = np.quantile(torso_points, [0.01, 0.99], axis=0)
if a.smooth_joints:
for side in ["Left", "Right"]:
for upper, lower in [
(side + "Arm", side + "ForeArm"),
(side + "Leg", side + "Shin"),
(side + "Shin", side + "Foot"),
]:
pivot = heads[lower]
direction = Vector(heads[lower] - heads[upper]).normalized()
length = np.linalg.norm(heads[lower] - heads[upper])
band = length * (
a.arm_blend_band if upper.endswith("Arm") else 0.35 if upper.endswith("Shin") else 0.55
)
for obj in meshes:
matrix = rig.matrix_world.inverted() @ obj.matrix_world
groups = {group.index: group.name for group in obj.vertex_groups}
for vertex in obj.data.vertices:
total = sum(
group.weight for group in vertex.groups if groups[group.group] in {upper, lower}
)
if total < 1e-6:
continue
point = matrix @ vertex.co
t = float(np.clip((np.dot(np.array(point) - pivot, direction) + band) / (2 * band), 0, 1))
t = t * t * (3 - 2 * t)
obj.vertex_groups[upper].add([vertex.index], total * (1 - t), "REPLACE")
obj.vertex_groups[lower].add([vertex.index], total * t, "REPLACE")
source_height = np.median(
pos[:, idx["HeadEnd"], 2] - np.minimum(pos[:, idx["LeftFoot"], 2], pos[:, idx["RightFoot"], 2])
)
target_height = (rig.data.bones["Head"].head_local - rig.data.bones["LeftFoot"].head_local).length
scale = target_height / max(source_height, 1e-6)
clearance_radius = target_height * 0.10
clearance_changes = 0
scene.frame_start = 1
scene.frame_end = len(pos)
for f in range(len(pos)):
scene.frame_set(f + 1)
rotations, targets = {}, {}
for bone in bones:
name = bone.name
parent = bone.parent.name if bone.parent else None
if parent:
targets[name] = targets[parent] + np.array(rotations[parent]) @ (heads[name] - heads[parent])
else:
delta = pos[f, idx["Hips"]] - pos[0, idx["Hips"]]
delta[:2] = 0
targets[name] = heads[name] + delta * scale
if name == "Head" or name.endswith("Hand"):
source_parent = "Neck1" if name == "Head" else name.replace("Hand", "ForeArm")
relative = source_rotations[f, idx[source_parent]].T @ source_rotations[f, idx[name]]
initial = source_rotations[0, idx[source_parent]].T @ source_rotations[0, idx[name]]
delta_rotation = Matrix((relative @ initial.T).tolist())
rotation = rotations[parent] @ delta_rotation
elif name in tips and name in idx and tips[name] in idx:
direction = rig.matrix_world.to_3x3().inverted() @ Vector(
pos[f, idx[tips[name]]] - pos[f, idx[name]]
)
if a.body_clearance and name.endswith(("Arm", "ForeArm")):
rest_direction = heads[tips[name]] - heads[name]
length = np.linalg.norm(rest_direction)
unit = np.array(direction.normalized())
end = targets[name] + unit * length
pad = clearance_radius
if (
body_low[1] - pad < end[1] < body_high[1] + pad
and body_low[2] - pad < end[2] < body_high[2] + pad
):
sign = 1 if name.startswith("Left") else -1
bound = body_high[0] + pad if sign > 0 else body_low[0] - pad
desired_x = (bound - targets[name][0]) / max(length, 1e-8)
if unit[0] * sign < desired_x * sign:
desired_x = float(np.clip(desired_x, -0.8, 0.8))
yz = unit[1:] / max(np.linalg.norm(unit[1:]), 1e-8)
direction = Vector((desired_x, *(yz * np.sqrt(1 - desired_x**2))))
clearance_changes += 1
if a.body_clearance and name.endswith(("Leg", "Shin")):
length = np.linalg.norm(heads[tips[name]] - heads[name])
unit = np.array(direction.normalized())
endpoint = targets[name] + unit * length
sign = 1 if name.startswith("Left") else -1
hip_x = targets["Hips"][0]
stance_x = hip_x + (heads[tips[name]][0] - heads["Hips"][0])
if (endpoint[0] - stance_x) * sign < 0:
desired_x = float(np.clip((stance_x - targets[name][0]) / max(length, 1e-8), -0.8, 0.8))
yz = unit[1:] / max(np.linalg.norm(unit[1:]), 1e-8)
direction = Vector((desired_x, *(yz * np.sqrt(1 - desired_x**2))))
clearance_changes += 1
rotation = (
(
Vector(heads[tips[name]] - heads[name])
if tips[name] in heads
else bone.tail_local - bone.head_local
)
.rotation_difference(Vector(direction))
.to_matrix()
)
else:
rotation = rotations[parent] if parent else Matrix.Identity(3)
rotations[name] = rotation
matrix = (rotation @ rest[name]).to_4x4()
matrix.translation = Vector(targets[name])
pb = rig.pose.bones[name]
pb.rotation_mode = "QUATERNION"
pb.matrix = matrix
bpy.context.view_layer.update()
for channel in ["location", "rotation_quaternion", "scale"]:
pb.keyframe_insert(channel, frame=f + 1)
# Preserve generated airborne phases while aligning source support contacts with the floor.
foot_ids = {}
for side, sign in [("Left", 1), ("Right", -1)]:
foot_ids[side] = []
for obj in meshes:
points = np.array([tuple(obj.matrix_world @ vertex.co) for vertex in obj.data.vertices])
foot_ids[side].append(np.where((points[:, 2] < 0.12) & (points[:, 0] * sign > 0))[0])
foot_minima = []
for frame in range(1, len(pos) + 1):
scene.frame_set(frame)
depsgraph = bpy.context.evaluated_depsgraph_get()
pair = []
for side in ["Left", "Right"]:
values = []
for obj, indices in zip(meshes, foot_ids[side]):
evaluated = obj.evaluated_get(depsgraph)
mesh = evaluated.to_mesh()
values.extend((evaluated.matrix_world @ mesh.vertices[int(i)].co).z for i in indices)
evaluated.to_mesh_clear()
if not values:
raise ValueError("Learned rig has no usable foot geometry for ground contact correction.")
pair.append(min(values))
foot_minima.append(pair)
foot_minima = np.array(foot_minima)
support = np.where(contacts, foot_minima, np.inf).min(axis=1)
support_frames = np.where(np.isfinite(support))[0]
if len(support_frames):
correction = np.interp(np.arange(len(pos)), support_frames, -support[support_frames])
else:
correction = np.zeros(len(pos))
correction = np.maximum(correction, -foot_minima.min(axis=1))
base_z = rig.location.z
for frame, offset in enumerate(correction, 1):
rig.location.z = base_z + float(offset)
rig.keyframe_insert("location", frame=frame)
rig.animation_data.action.name = "Generated_Motion"
scene.frame_set(1)
bpy.ops.object.select_all(action="DESELECT")
rig.select_set(True)
for obj in meshes:
obj.select_set(True)
bpy.context.view_layer.objects.active = rig
bpy.ops.export_scene.gltf(
filepath=str(out / "animation.glb"),
export_format="GLB",
use_selection=True,
export_animations=True,
export_skins=True,
export_frame_range=True,
)
bpy.ops.wm.save_as_mainfile(filepath=str(out / "animation.blend"))
(out / "retargeting.json").write_text(
json.dumps(
{
"frames": len(pos),
"fps": 30,
"method": "Generated SOMA joint directions on learned Mixamo skin; target proportions retained",
"limitations": [
"No synthesized finger or facial motion",
"Self intersections require audit review",
],
"root_translation": "vertical only; in-place preview",
"ground_correction": correction.tolist(),
"body_clearance_adjustments": clearance_changes,
"joint_weights_smoothed": a.smooth_joints,
"joint_blend_easing": "smoothstep",
"arm_blend_band_fraction": a.arm_blend_band if a.smooth_joints else None,
},
indent=2,
)
)