Add hand scaling from upstream, fix hand swapping when using dwpose

This commit is contained in:
kijai
2025-12-19 00:23:07 +02:00
parent 82380a5409
commit dad787e807
3 changed files with 37 additions and 20 deletions
+2 -2
View File
@@ -60,7 +60,7 @@ def solve_new_camera_params_central(three_d_points, focal_length, imshape, new_2
])
return K_final, m
return K_final, m, s
def solve_new_camera_params_down(three_d_points, focal_length, imshape, new_2d_points):
@@ -120,4 +120,4 @@ def solve_new_camera_params_down(three_d_points, focal_length, imshape, new_2d_p
])
return K_final, m
return K_final, m, s
+19 -6
View File
@@ -66,8 +66,11 @@ def intrinsic_matrix_from_field_of_view(imshape, fov_degrees:float =55): # nlf
[0, 0, 1],
])
def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, modified_intrinstics, height, width):
########## warning: 会改变body; shift 之后 body是不准的 ##########
def scale_around_center(points, center, dim, scale=1.0):
return (points[:, dim] - center[dim]) * scale + center[dim]
def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, modified_intrinstics, height, width, swap_hands=True, scale_hands=True, scale_x = 1.0, scale_y = 1.0):
########## warning: Will modify body; after shifting, the body is inaccurate ##########
for i in range(len(smpl_poses)):
persons_joints_list = smpl_poses[i]
poses_list = aligned_poses[i]
@@ -81,10 +84,13 @@ def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, mo
right_hand = poses_list["hands"][2 * person_idx]
left_hand = poses_list["hands"][2 * person_idx + 1]
candidate = poses_list["bodies"]["candidate"][person_idx]
# 注意,这里不是coco format
# Note: This is not COCO format
person_joint_15_2d_shift = p3d_single_p2d(person_joints[15], modified_intrinstics) - p3d_single_p2d(person_joints[15], ori_intrinstics) if person_joints[15, 2] > 0.01 else np.array([0.0, 0.0]) # face
person_joint_20_2d_shift = p3d_single_p2d(person_joints[20], modified_intrinstics) - p3d_single_p2d(person_joints[20], ori_intrinstics) if person_joints[20, 2] > 0.01 else np.array([0.0, 0.0]) # right hand
person_joint_21_2d_shift = p3d_single_p2d(person_joints[21], modified_intrinstics) - p3d_single_p2d(person_joints[21], ori_intrinstics) if person_joints[21, 2] > 0.01 else np.array([0.0, 0.0]) # left hand
person_joint_21_2d_shift = p3d_single_p2d(person_joints[20], modified_intrinstics) - p3d_single_p2d(person_joints[20], ori_intrinstics) if person_joints[20, 2] > 0.01 else np.array([0.0, 0.0]) # right hand
person_joint_20_2d_shift = p3d_single_p2d(person_joints[21], modified_intrinstics) - p3d_single_p2d(person_joints[21], ori_intrinstics) if person_joints[21, 2] > 0.01 else np.array([0.0, 0.0]) # left hand
if swap_hands:
person_joint_20_2d_shift, person_joint_21_2d_shift = person_joint_21_2d_shift, person_joint_20_2d_shift
face[:, 0] += person_joint_15_2d_shift[0] / width
face[:, 1] += person_joint_15_2d_shift[1] / height
@@ -95,13 +101,20 @@ def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, mo
candidate[:, 0] += person_joint_15_2d_shift[0] / width
candidate[:, 1] += person_joint_15_2d_shift[1] / height
scales = [scale_x, scale_y]
# apply camera scale around wrist (hand[0]).
if scale_hands:
for dim in [0,1]:
right_hand[:, dim] = scale_around_center(right_hand, right_hand[0, :], dim=dim, scale=scales[dim])
left_hand[:, dim] = scale_around_center(left_hand, left_hand[0, :], dim=dim, scale=scales[dim])
def get_single_pose_cylinder_specs(args):
"""Helper function for rendering a single pose, used for parallel processing."""
idx, pose, focal, princpt, height, width, colors, limb_seq, draw_seq = args
cylinder_specs = []
for joints3d in pose: # 多人
for joints3d in pose: # multiple persons
# Skip if None or not a valid tensor
if joints3d is None:
continue
+16 -12
View File
@@ -211,8 +211,9 @@ class PoseDetectionVitPoseToDWPose:
kp2ds = np.concatenate(kp2ds, 0)
pose_metas = load_pose_metas_from_kp2ds_seq(kp2ds, width=W, height=H)
dwposes = [aaposemeta_to_dwpose_scail(meta) for meta in pose_metas]
return (dwposes,)
swap_hands = True
out_dict = {"poses": dwposes, "swap_hands": swap_hands}
return out_dict,
class ConvertOpenPoseKeypointsToDWPose:
@@ -232,8 +233,9 @@ class ConvertOpenPoseKeypointsToDWPose:
DESCRIPTION = "Convert OpenPose format keypoints to DWPose format."
def process(self, keypoints, max_people=2):
return convert_openpose_to_target_format(keypoints, max_people=max_people),
swap_hands = False
out_dict = {"poses": convert_openpose_to_target_format(keypoints, max_people=max_people), "swap_hands": swap_hands}
return out_dict,
class RenderNLFPoses:
@@ -250,6 +252,7 @@ class RenderNLFPoses:
"draw_face": ("BOOLEAN", {"default": True, "tooltip": "Whether to draw face keypoints"}),
"draw_hands": ("BOOLEAN", {"default": True, "tooltip": "Whether to draw hand keypoints"}),
"render_device": (["gpu", "cpu", "opengl", "cuda", "vulkan", "metal"], {"default": "gpu", "tooltip": "Taichi device to use for rendering"}),
"scale_hands": ("BOOLEAN", {"default": True, "tooltip": "Whether to scale hand keypoints when aligning DW poses"}),
}
}
@@ -258,7 +261,7 @@ class RenderNLFPoses:
FUNCTION = "predict"
CATEGORY = "WanVideoWrapper"
def predict(self, nlf_poses, width, height, dw_poses=None, ref_dw_pose=None, draw_face=True, draw_hands=True, render_device="gpu"):
def predict(self, nlf_poses, width, height, dw_poses=None, ref_dw_pose=None, draw_face=True, draw_hands=True, render_device="gpu", scale_hands=True):
from .NLFPoseExtract.nlf_render import render_nlf_as_images, render_multi_nlf_as_images, shift_dwpose_according_to_nlf, process_data_to_COCO_format, intrinsic_matrix_from_field_of_view
from .NLFPoseExtract.align3d import solve_new_camera_params_central, solve_new_camera_params_down
@@ -280,7 +283,8 @@ class RenderNLFPoses:
else:
pose_input = nlf_poses
dw_pose_input = copy.deepcopy(dw_poses)
dw_pose_input = copy.deepcopy(dw_poses["poses"]) if dw_poses is not None else None
swap_hands = dw_poses.get("swap_hands", False) if dw_poses is not None else False
ori_camera_pose = intrinsic_matrix_from_field_of_view([height, width])
ori_focal = ori_camera_pose[0, 0]
@@ -288,7 +292,7 @@ class RenderNLFPoses:
num_people = dw_pose_input[0]['bodies']['candidate'].shape[0] if dw_poses is not None else 0
if dw_poses is not None and ref_dw_pose is not None and num_people == 1:
ref_dw_pose_input = copy.deepcopy(ref_dw_pose)
ref_dw_pose_input = copy.deepcopy(ref_dw_pose["poses"])
# Find the first valid pose
pose_3d_first_driving_frame = None
@@ -307,7 +311,7 @@ class RenderNLFPoses:
poses_2d_ref[:, 0] = poses_2d_ref[:, 0] * width
poses_2d_ref[:, 1] = poses_2d_ref[:, 1] * height
poses_2d_subset = ref_dw_pose[0]['bodies']['subset'][0][:14]
poses_2d_subset = ref_dw_pose_input[0]['bodies']['subset'][0][:14]
pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[:14]
valid_indices, valid_upper_indices, valid_lower_indices = [], [], []
@@ -327,14 +331,14 @@ class RenderNLFPoses:
pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[valid_indices]
if len(valid_lower_indices) >= 4:
new_camera_intrinsics, scale_m = solve_new_camera_params_down(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
new_camera_intrinsics, scale_m, scale_s = solve_new_camera_params_down(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
else:
new_camera_intrinsics, scale_m = solve_new_camera_params_central(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
new_camera_intrinsics, scale_m, scale_s = solve_new_camera_params_central(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
scale_face = scale_faces(list(dw_pose_input), list(ref_dw_pose)) # poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2
scale_face = scale_faces(list(dw_pose_input), list(ref_dw_pose_input)) # poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2
logging.info(f"Scale - m: {scale_m}, face: {scale_face}")
shift_dwpose_according_to_nlf(pose_input, dw_pose_input, ori_camera_pose, new_camera_intrinsics, height, width)
shift_dwpose_according_to_nlf(pose_input, dw_pose_input, ori_camera_pose, new_camera_intrinsics, height, width, swap_hands=swap_hands, scale_hands=scale_hands, scale_x=scale_m, scale_y=scale_m*scale_s)
intrinsic_matrix = new_camera_intrinsics
else: