diff --git a/NLFPoseExtract/align3d.py b/NLFPoseExtract/align3d.py index 20e6e06..069301c 100644 --- a/NLFPoseExtract/align3d.py +++ b/NLFPoseExtract/align3d.py @@ -60,7 +60,7 @@ def solve_new_camera_params_central(three_d_points, focal_length, imshape, new_2 ]) - return K_final, m + return K_final, m, s def solve_new_camera_params_down(three_d_points, focal_length, imshape, new_2d_points): @@ -120,4 +120,4 @@ def solve_new_camera_params_down(three_d_points, focal_length, imshape, new_2d_p ]) - return K_final, m + return K_final, m, s diff --git a/NLFPoseExtract/nlf_render.py b/NLFPoseExtract/nlf_render.py index 98eccd3..d31ec5d 100644 --- a/NLFPoseExtract/nlf_render.py +++ b/NLFPoseExtract/nlf_render.py @@ -66,8 +66,11 @@ def intrinsic_matrix_from_field_of_view(imshape, fov_degrees:float =55): # nlf [0, 0, 1], ]) -def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, modified_intrinstics, height, width): - ########## warning: 会改变body; shift 之后 body是不准的 ########## +def scale_around_center(points, center, dim, scale=1.0): + return (points[:, dim] - center[dim]) * scale + center[dim] + +def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, modified_intrinstics, height, width, swap_hands=True, scale_hands=True, scale_x = 1.0, scale_y = 1.0): + ########## warning: Will modify body; after shifting, the body is inaccurate ########## for i in range(len(smpl_poses)): persons_joints_list = smpl_poses[i] poses_list = aligned_poses[i] @@ -81,10 +84,13 @@ def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, mo right_hand = poses_list["hands"][2 * person_idx] left_hand = poses_list["hands"][2 * person_idx + 1] candidate = poses_list["bodies"]["candidate"][person_idx] - # 注意,这里不是coco format + # Note: This is not COCO format person_joint_15_2d_shift = p3d_single_p2d(person_joints[15], modified_intrinstics) - p3d_single_p2d(person_joints[15], ori_intrinstics) if person_joints[15, 2] > 0.01 else np.array([0.0, 0.0]) # face - person_joint_20_2d_shift = p3d_single_p2d(person_joints[20], modified_intrinstics) - p3d_single_p2d(person_joints[20], ori_intrinstics) if person_joints[20, 2] > 0.01 else np.array([0.0, 0.0]) # right hand - person_joint_21_2d_shift = p3d_single_p2d(person_joints[21], modified_intrinstics) - p3d_single_p2d(person_joints[21], ori_intrinstics) if person_joints[21, 2] > 0.01 else np.array([0.0, 0.0]) # left hand + person_joint_21_2d_shift = p3d_single_p2d(person_joints[20], modified_intrinstics) - p3d_single_p2d(person_joints[20], ori_intrinstics) if person_joints[20, 2] > 0.01 else np.array([0.0, 0.0]) # right hand + person_joint_20_2d_shift = p3d_single_p2d(person_joints[21], modified_intrinstics) - p3d_single_p2d(person_joints[21], ori_intrinstics) if person_joints[21, 2] > 0.01 else np.array([0.0, 0.0]) # left hand + + if swap_hands: + person_joint_20_2d_shift, person_joint_21_2d_shift = person_joint_21_2d_shift, person_joint_20_2d_shift face[:, 0] += person_joint_15_2d_shift[0] / width face[:, 1] += person_joint_15_2d_shift[1] / height @@ -95,13 +101,20 @@ def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, mo candidate[:, 0] += person_joint_15_2d_shift[0] / width candidate[:, 1] += person_joint_15_2d_shift[1] / height + scales = [scale_x, scale_y] + # apply camera scale around wrist (hand[0]). + if scale_hands: + for dim in [0,1]: + right_hand[:, dim] = scale_around_center(right_hand, right_hand[0, :], dim=dim, scale=scales[dim]) + left_hand[:, dim] = scale_around_center(left_hand, left_hand[0, :], dim=dim, scale=scales[dim]) + def get_single_pose_cylinder_specs(args): """Helper function for rendering a single pose, used for parallel processing.""" idx, pose, focal, princpt, height, width, colors, limb_seq, draw_seq = args cylinder_specs = [] - for joints3d in pose: # 多人 + for joints3d in pose: # multiple persons # Skip if None or not a valid tensor if joints3d is None: continue diff --git a/nodes.py b/nodes.py index fe3af64..a846091 100644 --- a/nodes.py +++ b/nodes.py @@ -211,8 +211,9 @@ class PoseDetectionVitPoseToDWPose: kp2ds = np.concatenate(kp2ds, 0) pose_metas = load_pose_metas_from_kp2ds_seq(kp2ds, width=W, height=H) dwposes = [aaposemeta_to_dwpose_scail(meta) for meta in pose_metas] - - return (dwposes,) + swap_hands = True + out_dict = {"poses": dwposes, "swap_hands": swap_hands} + return out_dict, class ConvertOpenPoseKeypointsToDWPose: @@ -232,8 +233,9 @@ class ConvertOpenPoseKeypointsToDWPose: DESCRIPTION = "Convert OpenPose format keypoints to DWPose format." def process(self, keypoints, max_people=2): - - return convert_openpose_to_target_format(keypoints, max_people=max_people), + swap_hands = False + out_dict = {"poses": convert_openpose_to_target_format(keypoints, max_people=max_people), "swap_hands": swap_hands} + return out_dict, class RenderNLFPoses: @@ -250,6 +252,7 @@ class RenderNLFPoses: "draw_face": ("BOOLEAN", {"default": True, "tooltip": "Whether to draw face keypoints"}), "draw_hands": ("BOOLEAN", {"default": True, "tooltip": "Whether to draw hand keypoints"}), "render_device": (["gpu", "cpu", "opengl", "cuda", "vulkan", "metal"], {"default": "gpu", "tooltip": "Taichi device to use for rendering"}), + "scale_hands": ("BOOLEAN", {"default": True, "tooltip": "Whether to scale hand keypoints when aligning DW poses"}), } } @@ -258,7 +261,7 @@ class RenderNLFPoses: FUNCTION = "predict" CATEGORY = "WanVideoWrapper" - def predict(self, nlf_poses, width, height, dw_poses=None, ref_dw_pose=None, draw_face=True, draw_hands=True, render_device="gpu"): + def predict(self, nlf_poses, width, height, dw_poses=None, ref_dw_pose=None, draw_face=True, draw_hands=True, render_device="gpu", scale_hands=True): from .NLFPoseExtract.nlf_render import render_nlf_as_images, render_multi_nlf_as_images, shift_dwpose_according_to_nlf, process_data_to_COCO_format, intrinsic_matrix_from_field_of_view from .NLFPoseExtract.align3d import solve_new_camera_params_central, solve_new_camera_params_down @@ -280,7 +283,8 @@ class RenderNLFPoses: else: pose_input = nlf_poses - dw_pose_input = copy.deepcopy(dw_poses) + dw_pose_input = copy.deepcopy(dw_poses["poses"]) if dw_poses is not None else None + swap_hands = dw_poses.get("swap_hands", False) if dw_poses is not None else False ori_camera_pose = intrinsic_matrix_from_field_of_view([height, width]) ori_focal = ori_camera_pose[0, 0] @@ -288,7 +292,7 @@ class RenderNLFPoses: num_people = dw_pose_input[0]['bodies']['candidate'].shape[0] if dw_poses is not None else 0 if dw_poses is not None and ref_dw_pose is not None and num_people == 1: - ref_dw_pose_input = copy.deepcopy(ref_dw_pose) + ref_dw_pose_input = copy.deepcopy(ref_dw_pose["poses"]) # Find the first valid pose pose_3d_first_driving_frame = None @@ -307,7 +311,7 @@ class RenderNLFPoses: poses_2d_ref[:, 0] = poses_2d_ref[:, 0] * width poses_2d_ref[:, 1] = poses_2d_ref[:, 1] * height - poses_2d_subset = ref_dw_pose[0]['bodies']['subset'][0][:14] + poses_2d_subset = ref_dw_pose_input[0]['bodies']['subset'][0][:14] pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[:14] valid_indices, valid_upper_indices, valid_lower_indices = [], [], [] @@ -327,14 +331,14 @@ class RenderNLFPoses: pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[valid_indices] if len(valid_lower_indices) >= 4: - new_camera_intrinsics, scale_m = solve_new_camera_params_down(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref) + new_camera_intrinsics, scale_m, scale_s = solve_new_camera_params_down(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref) else: - new_camera_intrinsics, scale_m = solve_new_camera_params_central(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref) + new_camera_intrinsics, scale_m, scale_s = solve_new_camera_params_central(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref) - scale_face = scale_faces(list(dw_pose_input), list(ref_dw_pose)) # poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2 + scale_face = scale_faces(list(dw_pose_input), list(ref_dw_pose_input)) # poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2 logging.info(f"Scale - m: {scale_m}, face: {scale_face}") - shift_dwpose_according_to_nlf(pose_input, dw_pose_input, ori_camera_pose, new_camera_intrinsics, height, width) + shift_dwpose_according_to_nlf(pose_input, dw_pose_input, ori_camera_pose, new_camera_intrinsics, height, width, swap_hands=swap_hands, scale_hands=scale_hands, scale_x=scale_m, scale_y=scale_m*scale_s) intrinsic_matrix = new_camera_intrinsics else: