This commit is contained in:
kijai
2025-12-14 19:16:28 +02:00
parent 05431879dd
commit 93059f0fe4
15 changed files with 3139 additions and 0 deletions
+13
View File
@@ -0,0 +1,13 @@
output/
*__pycache__/
samples*/
runs/
checkpoints/
master_ip
logs/
*.DS_Store
.idea
tools/
.vscode/
convert_*
*.pt
+123
View File
@@ -0,0 +1,123 @@
import numpy as np
from scipy.optimize import minimize
def solve_new_camera_params_central(three_d_points, focal_length, imshape, new_2d_points):
"""
Solve for new camera parameters by minimizing the error between the original 2D projection points and the new 2D projection points.
Args:
three_d_points (torch.Tensor): N*3 3D points
focal_length (float): Focal length of the original camera
imshape (tuple): Image size, e.g., [512, 896]
original_2d_points (torch.Tensor): N*2 original 2D projection points
new_2d_points (torch.Tensor): N*2 new 2D projection points
Returns:
m, n, p, q: Parameters in the new camera intrinsic matrix
"""
# Objective function: minimize the error between the original projection points and the new projection points
def objective(params):
m, s, p, q = params
# Construct the new camera intrinsic matrix
K_new = np.array([
[focal_length * m , 0, imshape[1] / 2 + p],
[0, focal_length * m * s, imshape[0] / 2 + q],
[0, 0, 1]
])
# Compute the new 2D projection points
new_projections = []
for point in three_d_points:
X, Y, Z = point
u = (K_new[0, 0] * X / Z) + K_new[0, 2]
v = (K_new[1, 1] * Y / Z) + K_new[1, 2]
new_projections.append([u, v])
new_projections = np.array(new_projections)
# Calculate the error between the original 2D projection points and the new projection points
# Special handling for the 0th projection point
error0 = np.sum((new_2d_points[:1] - new_projections[:1]) ** 2)
error = np.sum((new_2d_points[1:] - new_projections[1:]) ** 2)
return error0 * 8 + error
# Initialize parameters m, beta, p, q
initial_params = [1.0, 1.0, 0.0, 0.0] # Initial values
# Use least squares to solve for p, q
result = minimize(objective, initial_params, bounds=[(0.7, 1.4), (0.8, 1.15), (-imshape[1], imshape[1]), (-imshape[0], imshape[0])])
# Output the solution result
m, s, p, q = result.x
print(f"debug: solved camera params m={m}, s={s}, p={p}, q={q}")
K_final = np.array([
[focal_length * m, 0, imshape[1] / 2 + p],
[0, focal_length * m * s, imshape[0] / 2 + q],
[0, 0, 1]
])
return K_final, m
def solve_new_camera_params_down(three_d_points, focal_length, imshape, new_2d_points):
"""
Solve for new camera parameters by minimizing the error between the original 2D projection points and the new 2D projection points.
Args:
three_d_points (torch.Tensor): N*3 3D points
focal_length (float): Focal length of the original camera
imshape (tuple): Image size, e.g., [512, 896]
original_2d_points (torch.Tensor): N*2 original 2D projection points
new_2d_points (torch.Tensor): N*2 new 2D projection points
Returns:
m, n, p, q: Parameters in the new camera intrinsic matrix
"""
# Objective function: minimize the error between the original projection points and the new projection points
def objective(params):
m, s, p, q = params
# Construct the new camera intrinsic matrix
K_new = np.array([
[focal_length * m , 0, imshape[1] / 2 + p],
[0, focal_length * m * s, imshape[0] / 2 + q],
[0, 0, 1]
])
# Compute the new 2D projection points
new_projections = []
for point in three_d_points:
X, Y, Z = point
u = (K_new[0, 0] * X / Z) + K_new[0, 2]
v = (K_new[1, 1] * Y / Z) + K_new[1, 2]
new_projections.append([u, v])
new_projections = np.array(new_projections)
# Calculate the error between the original 2D projection points and the new projection points
# Special handling for the 0th projection point
error0 = np.sum((new_2d_points[:1] - new_projections[:1]) ** 2)
error = np.sum((new_2d_points[1:] - new_projections[1:]) ** 2)
return error0 + error * 4
# Initialize parameters m, beta, p, q
initial_params = [1.0, 1.0, 0.0, 0.0] # Initial values
# Use least squares to solve for p, q
result = minimize(objective, initial_params, bounds=[(0.7, 1.4), (0.8, 1.15), (-imshape[1], imshape[1]), (-imshape[0], imshape[0])])
# Output the solution result
m, s, p, q = result.x
print(f"debug: solved camera params m={m}, s={s}, p={p}, q={q}")
K_final = np.array([
[focal_length * m, 0, imshape[1] / 2 + p],
[0, focal_length * m * s, imshape[0] / 2 + q],
[0, 0, 1]
])
return K_final, m
+400
View File
@@ -0,0 +1,400 @@
import numpy as np
import torch
import os, platform, copy
if platform.system() == 'Linux':
if 'PYOPENGL_PLATFORM' not in os.environ:
os.environ['PYOPENGL_PLATFORM'] = 'egl'
elif platform.system() == 'Windows':
os.environ.pop('PYOPENGL_PLATFORM', None)
from ..render_3d.taichi_cylinder import render_whole
from ..pose_draw.draw_pose_utils import draw_pose_to_canvas_np
def p3d_single_p2d(points, intrinsic_matrix):
X, Y, Z = points[0], points[1], points[2]
u = (intrinsic_matrix[0, 0] * X / Z) + intrinsic_matrix[0, 2]
v = (intrinsic_matrix[1, 1] * Y / Z) + intrinsic_matrix[1, 2]
u_np = u.cpu().numpy()
v_np = v.cpu().numpy()
return np.array([u_np, v_np])
def process_data_to_COCO_format(joints):
"""Args:
joints: numpy array of shape (24, 2) or (24, 3)
Returns:
new_joints: numpy array of shape (17, 2) or (17, 3)
"""
if joints.ndim != 2:
raise ValueError(f"Expected shape (24,2) or (24,3), got {joints.shape}")
dim = joints.shape[1] # 2D or 3D
mapping = {
15: 0, # head
12: 1, # neck
17: 2, # left shoulder
16: 5, # right shoulder
19: 3, # left elbow
18: 6, # right elbow
21: 4, # left hand
20: 7, # right hand
2: 8, # left pelvis
1: 11, # right pelvis
5: 9, # left knee
4: 12, # right knee
8: 10, # left feet
7: 13, # right feet
}
new_joints = np.zeros((18, dim), dtype=joints.dtype)
for src, dst in mapping.items():
new_joints[dst] = joints[src]
return new_joints
def intrinsic_matrix_from_field_of_view(imshape, fov_degrees:float =55): # nlf default fov_degrees 55
imshape = np.array(imshape)
fov_radians = fov_degrees * np.array(np.pi / 180)
larger_side = np.max(imshape)
focal_length = larger_side / (np.tan(fov_radians / 2) * 2)
# intrinsic_matrix 3*3
return np.array([
[focal_length, 0, imshape[1] / 2],
[0, focal_length, imshape[0] / 2],
[0, 0, 1],
])
def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, modified_intrinstics, height, width):
########## warning: 会改变body; shift 之后 body是不准的 ##########
for i in range(len(smpl_poses)):
persons_joints_list = smpl_poses[i]
poses_list = aligned_poses[i]
# 对里面每一个人,取关节并进行变形;并且修改2d;如果3d不存在,把2d的手/脸也去掉
for person_idx, person_joints in enumerate(persons_joints_list):
face = poses_list["faces"][person_idx]
right_hand = poses_list["hands"][2 * person_idx]
left_hand = poses_list["hands"][2 * person_idx + 1]
candidate = poses_list["bodies"]["candidate"][person_idx]
# 注意,这里不是coco format
person_joint_15_2d_shift = p3d_single_p2d(person_joints[15], modified_intrinstics) - p3d_single_p2d(person_joints[15], ori_intrinstics) if person_joints[15, 2] > 0.01 else np.array([0.0, 0.0]) # face
person_joint_20_2d_shift = p3d_single_p2d(person_joints[20], modified_intrinstics) - p3d_single_p2d(person_joints[20], ori_intrinstics) if person_joints[20, 2] > 0.01 else np.array([0.0, 0.0]) # right hand
person_joint_21_2d_shift = p3d_single_p2d(person_joints[21], modified_intrinstics) - p3d_single_p2d(person_joints[21], ori_intrinstics) if person_joints[21, 2] > 0.01 else np.array([0.0, 0.0]) # left hand
face[:, 0] += person_joint_15_2d_shift[0] / width
face[:, 1] += person_joint_15_2d_shift[1] / height
right_hand[:, 0] += person_joint_20_2d_shift[0] / width
right_hand[:, 1] += person_joint_20_2d_shift[1] / height
left_hand[:, 0] += person_joint_21_2d_shift[0] / width
left_hand[:, 1] += person_joint_21_2d_shift[1] / height
candidate[:, 0] += person_joint_15_2d_shift[0] / width
candidate[:, 1] += person_joint_15_2d_shift[1] / height
def get_single_pose_cylinder_specs(args):
"""Helper function for rendering a single pose, used for parallel processing."""
idx, pose, focal, princpt, height, width, colors, limb_seq, draw_seq = args
cylinder_specs = []
for joints3d in pose: # 多人
joints3d = joints3d.cpu().numpy()
joints3d = process_data_to_COCO_format(joints3d)
for line_idx in draw_seq:
line = limb_seq[line_idx]
start, end = line[0], line[1]
if np.sum(joints3d[start]) == 0 or np.sum(joints3d[end]) == 0:
continue
else:
cylinder_specs.append((joints3d[start], joints3d[end], colors[line_idx]))
return cylinder_specs
def collect_smpl_poses(data):
uncollected_smpl_poses = [item['nlfpose'] for item in data]
smpl_poses = [[] for _ in range(len(uncollected_smpl_poses))]
for frame_idx in range(len(uncollected_smpl_poses)):
for person_idx in range(len(uncollected_smpl_poses[frame_idx])): # 每个人(每个bbox)只给出一个pose
if len(uncollected_smpl_poses[frame_idx][person_idx]) > 0: # 有返回的骨骼
smpl_poses[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0])
else:
smpl_poses[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32)) # 没有检测到人,就放一个全0的
return smpl_poses
def collect_smpl_poses_samurai(data):
uncollected_smpl_poses = [item['nlfpose'] for item in data]
smpl_poses_first = [[] for _ in range(len(uncollected_smpl_poses))]
smpl_poses_second = [[] for _ in range(len(uncollected_smpl_poses))]
for frame_idx in range(len(uncollected_smpl_poses)):
for person_idx in range(len(uncollected_smpl_poses[frame_idx])): # 每个人(每个bbox)只给出一个pose
if len(uncollected_smpl_poses[frame_idx][person_idx]) > 0: # 有返回的骨骼
if person_idx == 0:
smpl_poses_first[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0])
elif person_idx == 1:
smpl_poses_second[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0])
else:
if person_idx == 0:
smpl_poses_first[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32)) # 没有检测到人,就放一个全0的
elif person_idx == 1:
smpl_poses_second[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32))
return smpl_poses_first, smpl_poses_second
def render_nlf_as_images(smpl_poses, dw_poses, height, width, video_length, intrinsic_matrix=None, draw_2d=True):
""" return a list of images """
base_colors_255_dict = {
# Warm Colors for Right Side (R.) - Red, Orange, Yellow
"Red": [255, 0, 0],
"Orange": [255, 85, 0],
"Golden Orange": [255, 170, 0],
"Yellow": [255, 240, 0],
"Yellow-Green": [180, 255, 0],
# Cool Colors for Left Side (L.) - Green, Blue, Purple
"Bright Green": [0, 255, 0],
"Light Green-Blue": [0, 255, 85],
"Aqua": [0, 255, 170],
"Cyan": [0, 255, 255],
"Sky Blue": [0, 170, 255],
"Medium Blue": [0, 85, 255],
"Pure Blue": [0, 0, 255],
"Purple-Blue": [85, 0, 255],
"Medium Purple": [170, 0, 255],
# Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears)
"Grey": [150, 150, 150],
"Pink-Magenta": [255, 0, 170],
"Dark Pink": [255, 0, 85],
"Violet": [100, 0, 255],
"Dark Violet": [50, 0, 255],
}
ordered_colors_255 = [
base_colors_255_dict["Red"], # Neck -> R. Shoulder (Red)
base_colors_255_dict["Cyan"], # Neck -> L. Shoulder (Cyan)
base_colors_255_dict["Orange"], # R. Shoulder -> R. Elbow (Orange)
base_colors_255_dict["Golden Orange"], # R. Elbow -> R. Wrist (Golden Orange)
base_colors_255_dict["Sky Blue"], # L. Shoulder -> L. Elbow (Sky Blue)
base_colors_255_dict["Medium Blue"], # L. Elbow -> L. Wrist (Medium Blue)
base_colors_255_dict["Yellow-Green"], # Neck -> R. Hip ( Yellow-Green)
base_colors_255_dict["Bright Green"], # R. Hip -> R. Knee (Bright Green - transitioning warm to cool spectrum)
base_colors_255_dict["Light Green-Blue"], # R. Knee -> R. Ankle (Light Green-Blue - transitioning)
base_colors_255_dict["Pure Blue"], # Neck -> L. Hip (Pure Blue)
base_colors_255_dict["Purple-Blue"], # L. Hip -> L. Knee (Purple-Blue)
base_colors_255_dict["Medium Purple"], # L. Knee -> L. Ankle (Medium Purple)
base_colors_255_dict["Grey"], # Neck -> Nose (Grey)
base_colors_255_dict["Pink-Magenta"], # Nose -> R. Eye (Pink/Magenta)
base_colors_255_dict["Dark Violet"], # R. Eye -> R. Ear (Dark Pink)
base_colors_255_dict["Pink-Magenta"], # Nose -> L. Eye (Violet)
base_colors_255_dict["Dark Violet"], # L. Eye -> L. Ear (Dark Violet)
]
limb_seq = [
[1, 2], # 0 Neck -> R. Shoulder
[1, 5], # 1 Neck -> L. Shoulder
[2, 3], # 2 R. Shoulder -> R. Elbow
[3, 4], # 3 R. Elbow -> R. Wrist
[5, 6], # 4 L. Shoulder -> L. Elbow
[6, 7], # 5 L. Elbow -> L. Wrist
[1, 8], # 6 Neck -> R. Hip
[8, 9], # 7 R. Hip -> R. Knee
[9, 10], # 8 R. Knee -> R. Ankle
[1, 11], # 9 Neck -> L. Hip
[11, 12], # 10 L. Hip -> L. Knee
[12, 13], # 11 L. Knee -> L. Ankle
[1, 0], # 12 Neck -> Nose
[0, 14], # 13 Nose -> R. Eye
[14, 16], # 14 R. Eye -> R. Ear
[0, 15], # 15 Nose -> L. Eye
[15, 17], # 16 L. Eye -> L. Ear
]
draw_seq = [0, 2, 3, # Neck -> R. Shoulder -> R. Elbow -> R. Wrist
1, 4, 5, # Neck -> L. Shoulder -> L. Elbow -> L. Wrist
6, 7, 8, # Neck -> R. Hip -> R. Knee -> R. Ankle
9, 10, 11, # Neck -> L. Hip -> L. Knee -> L. Ankle
12, # Neck -> Nose
13, 14, # Nose -> R. Eye -> R. Ear
15, 16, # Nose -> L. Eye -> L. Ear
] # Expanding outward from the proximal end
colors = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255]
if dw_poses is not None:
aligned_poses = copy.deepcopy(dw_poses)
if intrinsic_matrix is None:
intrinsic_matrix = intrinsic_matrix_from_field_of_view((height, width))
focal_x = intrinsic_matrix[0,0]
focal_y = intrinsic_matrix[1,1]
princpt = (intrinsic_matrix[0,2], intrinsic_matrix[1,2]) # (cx, cy)
# obtain cylinder_specs for each frame
cylinder_specs_list = []
for i in range(video_length):
cylinder_specs = get_single_pose_cylinder_specs((i, smpl_poses[i], None, None, None, None, colors, limb_seq, draw_seq))
cylinder_specs_list.append(cylinder_specs)
frames_np_rgba = render_whole(cylinder_specs_list, H=height, W=width, fx=focal_x, fy=focal_y, cx=princpt[0], cy=princpt[1])
if dw_poses is not None and draw_2d:
canvas_2d = draw_pose_to_canvas_np(aligned_poses, pool=None, H=height, W=width, reshape_scale=0, show_feet_flag=False, show_body_flag=False, show_cheek_flag=True, dw_hand=True)
for i in range(len(frames_np_rgba)):
frame_img = frames_np_rgba[i]
canvas_img = canvas_2d[i]
mask = canvas_img != 0
frame_img[:, :, :3][mask] = canvas_img[mask]
frames_np_rgba[i] = frame_img
return frames_np_rgba
def render_multi_nlf_as_images(data, dw_poses, intrinsic_matrix=None, draw_2d=True):
""" return a list of images """
height, width = data[0]['video_height'], data[0]['video_width']
video_length = len(data)
second_person_base_colors_255_dict = {
# Warm Colors for Right Side (R.) - Red, Orange, Yellow
"Red": [255, 20, 20],
"Orange": [255, 60, 0],
"Golden Orange": [255, 110, 0],
"Yellow": [255, 200, 0],
"Yellow-Green": [160, 255, 40],
# Cool Colors for Left Side (L.) - Green, Blue, Purple
"Bright Green": [0, 255, 50],
"Light Green-Blue": [0, 255, 100],
"Aqua": [0, 255, 200],
"Cyan": [0, 230, 255],
"Sky Blue": [0, 130, 255],
"Medium Blue": [0, 70, 255],
"Pure Blue": [0, 0, 255],
"Purple-Blue": [80, 0, 255],
"Medium Purple": [160, 0, 255],
# Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears)
"Grey": [130, 130, 130],
"Pink-Magenta": [255, 0, 150],
"Dark Pink": [255, 0, 100],
"Violet": [120, 0, 255],
"Dark Violet": [60, 0, 255],
}
first_person_base_colors_255_dict = {
# Warm Colors for Right Side (R.) - Red, Orange, Yellow
"Red": [255, 150, 150],
"Orange": [255, 180, 140],
"Golden Orange": [255, 215, 150],
"Yellow": [255, 240, 170],
"Yellow-Green": [200, 255, 100],
# Cool Colors for Left Side (L.) - Green, Blue, Purple
"Bright Green": [100, 255, 100],
"Light Green-Blue": [140, 255, 180],
"Aqua": [150, 240, 200],
"Cyan": [180, 230, 240],
"Sky Blue": [160, 200, 255],
"Medium Blue": [100, 120, 255],
"Pure Blue": [120, 140, 255],
"Purple-Blue": [180, 90, 255],
"Medium Purple": [190, 120, 255],
# Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears)
"Grey": [210, 210, 210],
"Pink-Magenta": [255, 120, 200],
"Dark Pink": [255, 150, 180],
"Violet": [200, 90, 255],
"Dark Violet": [130, 80, 255],
}
base_colors_255_dict_list = [first_person_base_colors_255_dict, second_person_base_colors_255_dict]
ordered_colors_255_list = [[
base_colors_255_dict["Red"], # Neck -> R. Shoulder (Red)
base_colors_255_dict["Cyan"], # Neck -> L. Shoulder (Cyan)
base_colors_255_dict["Orange"], # R. Shoulder -> R. Elbow (Orange)
base_colors_255_dict["Golden Orange"], # R. Elbow -> R. Wrist (Golden Orange)
base_colors_255_dict["Sky Blue"], # L. Shoulder -> L. Elbow (Sky Blue)
base_colors_255_dict["Medium Blue"], # L. Elbow -> L. Wrist (Medium Blue)
base_colors_255_dict["Yellow-Green"], # Neck -> R. Hip ( Yellow-Green)
base_colors_255_dict["Bright Green"], # R. Hip -> R. Knee (Bright Green - transitioning warm to cool spectrum)
base_colors_255_dict["Light Green-Blue"], # R. Knee -> R. Ankle (Light Green-Blue - transitioning)
base_colors_255_dict["Pure Blue"], # Neck -> L. Hip (Pure Blue)
base_colors_255_dict["Purple-Blue"], # L. Hip -> L. Knee (Purple-Blue)
base_colors_255_dict["Medium Purple"], # L. Knee -> L. Ankle (Medium Purple)
base_colors_255_dict["Grey"], # Neck -> Nose (Grey)
base_colors_255_dict["Pink-Magenta"], # Nose -> R. Eye (Pink/Magenta)
base_colors_255_dict["Dark Violet"], # R. Eye -> R. Ear (Dark Pink)
base_colors_255_dict["Pink-Magenta"], # Nose -> L. Eye (Violet)
base_colors_255_dict["Dark Violet"], # L. Eye -> L. Ear (Dark Violet)
] for base_colors_255_dict in base_colors_255_dict_list]
limb_seq = [
[1, 2], # 0 Neck -> R. Shoulder
[1, 5], # 1 Neck -> L. Shoulder
[2, 3], # 2 R. Shoulder -> R. Elbow
[3, 4], # 3 R. Elbow -> R. Wrist
[5, 6], # 4 L. Shoulder -> L. Elbow
[6, 7], # 5 L. Elbow -> L. Wrist
[1, 8], # 6 Neck -> R. Hip
[8, 9], # 7 R. Hip -> R. Knee
[9, 10], # 8 R. Knee -> R. Ankle
[1, 11], # 9 Neck -> L. Hip
[11, 12], # 10 L. Hip -> L. Knee
[12, 13], # 11 L. Knee -> L. Ankle
[1, 0], # 12 Neck -> Nose
[0, 14], # 13 Nose -> R. Eye
[14, 16], # 14 R. Eye -> R. Ear
[0, 15], # 15 Nose -> L. Eye
[15, 17], # 16 L. Eye -> L. Ear
]
draw_seq = [0, 2, 3, # Neck -> R. Shoulder -> R. Elbow -> R. Wrist
1, 4, 5, # Neck -> L. Shoulder -> L. Elbow -> L. Wrist
6, 7, 8, # Neck -> R. Hip -> R. Knee -> R. Ankle
9, 10, 11, # Neck -> L. Hip -> L. Knee -> L. Ankle
12, # Neck -> Nose
13, 14, # Nose -> R. Eye -> R. Ear
15, 16, # Nose -> L. Eye -> L. Ear
] # Expanding outward from the proximal end
colors_first = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255_list[0]]
colors_second = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255_list[1]]
smpl_poses_first, smpl_poses_second = collect_smpl_poses_samurai(data)
if intrinsic_matrix is None:
intrinsic_matrix = intrinsic_matrix_from_field_of_view((height, width))
focal_x = intrinsic_matrix[0,0]
focal_y = intrinsic_matrix[1,1]
princpt = (intrinsic_matrix[0,2], intrinsic_matrix[1,2]) # (cx, cy)
# obtain cylinder_specs for each frame
cylinder_specs_list = []
for i in range(video_length):
cylinder_specs_first = get_single_pose_cylinder_specs((i, smpl_poses_first[i], None, None, None, None, colors_first, limb_seq, draw_seq))
cylinder_specs_second = get_single_pose_cylinder_specs((i, smpl_poses_second[i], None, None, None, None, colors_second, limb_seq, draw_seq))
cylinder_specs = cylinder_specs_first + cylinder_specs_second
cylinder_specs_list.append(cylinder_specs)
frames_np_rgba = render_whole(cylinder_specs_list, H=height, W=width, fx=focal_x, fy=focal_y, cx=princpt[0], cy=princpt[1])
if dw_poses is not None and draw_2d:
aligned_poses = copy.deepcopy(dw_poses)
canvas_2d = draw_pose_to_canvas_np(aligned_poses, pool=None, H=height, W=width, reshape_scale=0, show_feet_flag=False, show_body_flag=False, show_cheek_flag=True, dw_hand=True)
for i in range(len(frames_np_rgba)):
frame_img = frames_np_rgba[i]
canvas_img = canvas_2d[i]
mask = canvas_img != 0
frame_img[:, :, :3][mask] = canvas_img[mask]
frames_np_rgba[i] = frame_img
return frames_np_rgba
+3
View File
@@ -0,0 +1,3 @@
from .nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
@@ -0,0 +1,825 @@
{
"id": "c6e410bc-5e2c-460b-ae81-c91b6094fbb1",
"revision": 0,
"last_node_id": 381,
"last_link_id": 706,
"nodes": [
{
"id": 379,
"type": "VHS_LoadVideo",
"pos": [
-728.1399554193044,
-1903.2296105502905
],
"size": [
255.8291015625,
743.5855352640658
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"shape": 7,
"type": "VHS_BatchManager",
"link": null
},
{
"name": "vae",
"shape": 7,
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
684
]
},
{
"name": "frame_count",
"type": "INT",
"links": []
},
{
"name": "audio",
"type": "AUDIO",
"links": null
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "8550981384301e9bc5bfea83e5c2c75258102593",
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "vid.mp4",
"force_rate": 0,
"custom_width": 0,
"custom_height": 0,
"frame_load_cap": 81,
"skip_first_frames": 0,
"select_every_nth": 2,
"format": "AnimateDiff",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "vid.mp4",
"type": "input",
"format": "video/mp4",
"force_rate": 0,
"custom_width": 0,
"custom_height": 0,
"frame_load_cap": 81,
"skip_first_frames": 0,
"select_every_nth": 2
}
}
}
},
{
"id": 358,
"type": "OnnxDetectionModelLoader",
"pos": [
-731.2840742077292,
-2127.3788215536956
],
"size": [
304.0484375,
106
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "model",
"type": "POSEMODEL",
"links": [
677,
680
]
}
],
"properties": {
"cnr_id": "ComfyUI-WanAnimatePreprocess",
"ver": "65502e208ae89231619e3def41bc8bafe6fc4f1e",
"Node name for S&R": "OnnxDetectionModelLoader"
},
"widgets_values": [
"vitpose-l-wholebody.onnx",
"onnx\\yolov10m.onnx",
"CUDAExecutionProvider"
]
},
{
"id": 361,
"type": "NLFPredict",
"pos": [
485.2315622592827,
-2157.1952980266337
],
"size": [
157.564453125,
46
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "NLFMODEL",
"link": 627
},
{
"name": "images",
"type": "IMAGE",
"link": 701
}
],
"outputs": [
{
"name": "pose_results",
"type": "NLFPRED",
"links": [
656
]
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "056d8ad96ffa5a223a8cd88c900a573a8d450e22",
"Node name for S&R": "NLFPredict"
},
"widgets_values": []
},
{
"id": 362,
"type": "DownloadAndLoadNLFModel",
"pos": [
-279.5657072597091,
-2231.190892106357
],
"size": [
606.9488335754912,
82
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "nlf_model",
"type": "NLFMODEL",
"links": [
627
]
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "056d8ad96ffa5a223a8cd88c900a573a8d450e22",
"Node name for S&R": "DownloadAndLoadNLFModel"
},
"widgets_values": [
"https://github.com/isarandi/nlf/releases/download/v0.3.2/nlf_l_multi_0.3.2.torchscript",
true
]
},
{
"id": 359,
"type": "LoadImage",
"pos": [
-369.63293579008285,
-1510.3591727642465
],
"size": [
282.798828125,
314
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
698
]
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"cnr_id": "comfy-core",
"ver": "0.4.0",
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"anya.jpg",
"image"
]
},
{
"id": 381,
"type": "Reroute",
"pos": [
144.49613132898912,
-2052.387551799464
],
"size": [
75,
26
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "",
"type": "*",
"link": 700
}
],
"outputs": [
{
"name": "",
"type": "IMAGE",
"links": [
701,
702
]
}
],
"properties": {
"showOutputText": false,
"horizontal": false
}
},
{
"id": 380,
"type": "VHS_VideoCombine",
"pos": [
1061.113369865757,
-2147.425691347535
],
"size": [
329.65692831178876,
889.8996245456303
],
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 685
},
{
"name": "audio",
"shape": 7,
"type": "AUDIO",
"link": null
},
{
"name": "meta_batch",
"shape": 7,
"type": "VHS_BatchManager",
"link": null
},
{
"name": "vae",
"shape": 7,
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "8550981384301e9bc5bfea83e5c2c75258102593",
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 16,
"loop_count": 0,
"filename_prefix": "SCAIL_pose",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": false,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "SCAIL_pose_00030.mp4",
"subfolder": "",
"type": "temp",
"format": "video/h264-mp4",
"frame_rate": 16,
"workflow": "SCAIL_pose_00030.png",
"fullpath": "N:\\AI\\ComfyUI\\temp\\SCAIL_pose_00030.mp4"
}
}
}
},
{
"id": 376,
"type": "PoseDetectionVitPoseToDWPose",
"pos": [
330.07991567045417,
-1961.6148428238132
],
"size": [
271.9595703125,
46
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "vitpose_model",
"type": "POSEMODEL",
"link": 677
},
{
"name": "images",
"type": "IMAGE",
"link": 702
}
],
"outputs": [
{
"name": "dw_poses",
"type": "DWPOSES",
"links": [
704
]
}
],
"properties": {
"Node name for S&R": "PoseDetectionVitPoseToDWPose"
},
"widgets_values": []
},
{
"id": 370,
"type": "RenderNLFPoses",
"pos": [
747.8706167185883,
-2157.040108369948
],
"size": [
270,
146
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "nlf_poses",
"type": "NLFPRED",
"link": 656
},
{
"name": "dw_poses",
"shape": 7,
"type": "DWPOSES",
"link": 704
},
{
"name": "ref_dw_pose",
"shape": 7,
"type": "DWPOSES",
"link": 706
},
{
"name": "width",
"type": "INT",
"widget": {
"name": "width"
},
"link": 688
},
{
"name": "height",
"type": "INT",
"widget": {
"name": "height"
},
"link": 689
}
],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": [
685
]
},
{
"name": "mask",
"type": "MASK",
"links": []
}
],
"properties": {
"Node name for S&R": "RenderNLFPoses"
},
"widgets_values": [
512,
896
]
},
{
"id": 372,
"type": "ImageResizeKJv2",
"pos": [
-379.6838809890405,
-1915.7782190865344
],
"size": [
270,
336
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 684
},
{
"name": "mask",
"shape": 7,
"type": "MASK",
"link": null
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
700
]
},
{
"name": "width",
"type": "INT",
"links": [
688,
695
]
},
{
"name": "height",
"type": "INT",
"links": [
689,
696
]
},
{
"name": "mask",
"type": "MASK",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-kjnodes",
"ver": "1b9af2322c009e0171f1c7c2f8457ca69afa1ec0",
"Node name for S&R": "ImageResizeKJv2"
},
"widgets_values": [
512,
896,
"bilinear",
"crop",
"0, 0, 0",
"center",
2,
"cpu",
"<tr><td>Output: </td><td><b>81</b> x <b>512</b> x <b>896 | 425.25MB</b></td></tr>"
]
},
{
"id": 374,
"type": "ImageResizeKJv2",
"pos": [
-46.738037813889264,
-1635.669074975477
],
"size": [
270,
336
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 698
},
{
"name": "mask",
"shape": 7,
"type": "MASK",
"link": null
},
{
"name": "width",
"type": "INT",
"widget": {
"name": "width"
},
"link": 695
},
{
"name": "height",
"type": "INT",
"widget": {
"name": "height"
},
"link": 696
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
682
]
},
{
"name": "width",
"type": "INT",
"links": null
},
{
"name": "height",
"type": "INT",
"links": null
},
{
"name": "mask",
"type": "MASK",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-kjnodes",
"ver": "1b9af2322c009e0171f1c7c2f8457ca69afa1ec0",
"Node name for S&R": "ImageResizeKJv2"
},
"widgets_values": [
512,
896,
"bilinear",
"pad",
"0, 0, 0",
"center",
2,
"cpu",
"<tr><td>Output: </td><td><b>1</b> x <b>512</b> x <b>896 | 5.25MB</b></td></tr>"
]
},
{
"id": 377,
"type": "PoseDetectionVitPoseToDWPose",
"pos": [
361.20948481109036,
-1818.3087110204879
],
"size": [
271.9595703125,
46
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "vitpose_model",
"type": "POSEMODEL",
"link": 680
},
{
"name": "images",
"type": "IMAGE",
"link": 682
}
],
"outputs": [
{
"name": "dw_poses",
"type": "DWPOSES",
"links": [
706
]
}
],
"properties": {
"Node name for S&R": "PoseDetectionVitPoseToDWPose"
},
"widgets_values": []
}
],
"links": [
[
627,
362,
0,
361,
0,
"NLFMODEL"
],
[
656,
361,
0,
370,
0,
"NLFPRED"
],
[
677,
358,
0,
376,
0,
"POSEMODEL"
],
[
680,
358,
0,
377,
0,
"POSEMODEL"
],
[
682,
374,
0,
377,
1,
"IMAGE"
],
[
684,
379,
0,
372,
0,
"IMAGE"
],
[
685,
370,
0,
380,
0,
"IMAGE"
],
[
688,
372,
1,
370,
3,
"INT"
],
[
689,
372,
2,
370,
4,
"INT"
],
[
695,
372,
1,
374,
2,
"INT"
],
[
696,
372,
2,
374,
3,
"INT"
],
[
698,
359,
0,
374,
0,
"IMAGE"
],
[
700,
372,
0,
381,
0,
"IMAGE"
],
[
701,
381,
0,
361,
1,
"IMAGE"
],
[
702,
381,
0,
376,
1,
"IMAGE"
],
[
704,
376,
0,
370,
1,
"DWPOSES"
],
[
706,
377,
0,
370,
2,
"DWPOSES"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.8140274938684753,
"offset": [
1341.9401489123059,
2475.387199967974
]
},
"frontendVersion": "1.35.3",
"workflowRendererVersion": "LG",
"node_versions": {
"ComfyUI-WanVideoWrapper": "5a2383621a05825d0d0437781afcb8552d9590fd",
"comfy-core": "0.3.26",
"ComfyUI-VideoHelperSuite": "0a75c7958fe320efcb052f1d9f8451fd20c730a8"
},
"VHS_latentpreview": true,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
+243
View File
@@ -0,0 +1,243 @@
import os
import torch
from tqdm import tqdm
import numpy as np
import folder_paths
import cv2
import logging
import copy
script_directory = os.path.dirname(os.path.abspath(__file__))
from comfy import model_management as mm
from comfy.utils import ProgressBar
device = mm.get_torch_device()
offload_device = mm.unet_offload_device()
folder_paths.add_model_folder_path("detection", os.path.join(folder_paths.models_dir, "detection"))
from .vitpose_utils.utils import bbox_from_detector, crop, load_pose_metas_from_kp2ds_seq, aaposemeta_to_dwpose_scail
def scale_faces(poses, pose_2d_ref):
# Input: two lists of dict, poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2
# Scale the facial keypoints in poses according to the center point of the face
# That is: calculate the distance from the center point (idx: 30) to other facial keypoints in ref,
# and the same for poses, then get scale_n as the ratio
# Clamp scale_n to the range 0.8-1.5, then apply it to poses
# Note: poses are modified in place
ref = pose_2d_ref[0]
pose_0 = poses[0]
face_0 = pose_0['faces'] # shape: (1, 68, 2)
face_ref = ref['faces']
# Extract numpy arrays
face_0 = np.array(face_0[0]) # (68, 2)
face_ref = np.array(face_ref[0])
# Center point (nose tip or face center)
center_idx = 30
center_0 = face_0[center_idx]
center_ref = face_ref[center_idx]
# Calculate distance to center point
dist = np.linalg.norm(face_0 - center_0, axis=1)
dist_ref = np.linalg.norm(face_ref - center_ref, axis=1)
# Avoid the 0 distance of the center point itself
dist = np.delete(dist, center_idx)
dist_ref = np.delete(dist_ref, center_idx)
mean_dist = np.mean(dist)
mean_dist_ref = np.mean(dist_ref)
if mean_dist < 1e-6:
scale_n = 1.0
else:
scale_n = mean_dist_ref / mean_dist
# Clamp to [0.8, 1.5]
scale_n = np.clip(scale_n, 0.8, 1.5)
for i, pose in enumerate(poses):
face = pose['faces']
# Extract numpy array
face = np.array(face[0]) # (68, 2)
center = face[center_idx]
scaled_face = (face - center) * scale_n + center
poses[i]['faces'][0] = scaled_face
body = pose['bodies']
candidate = body['candidate']
candidate_np = np.array(candidate[0]) # (14, 2)
body_center = candidate_np[0]
scaled_candidate = (candidate_np - body_center) * scale_n + body_center
poses[i]['bodies']['candidate'][0] = scaled_candidate
# In-place modification
pose['faces'][0] = scaled_face
return scale_n
class PoseDetectionVitPoseToDWPose:
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"vitpose_model": ("POSEMODEL",),
"images": ("IMAGE",),
},
}
RETURN_TYPES = ("DWPOSES",)
RETURN_NAMES = ("dw_poses",)
FUNCTION = "process"
CATEGORY = "WanAnimatePreprocess"
DESCRIPTION = "ViTPose to DWPose format pose detection node."
def process(self, vitpose_model, images):
detector = vitpose_model["yolo"]
pose_model = vitpose_model["vitpose"]
B, H, W, C = images.shape
shape = np.array([H, W])[None]
images_np = images.numpy()
IMG_NORM_MEAN = np.array([0.485, 0.456, 0.406])
IMG_NORM_STD = np.array([0.229, 0.224, 0.225])
input_resolution=(256, 192)
rescale = 1.25
detector.reinit()
pose_model.reinit()
comfy_pbar = ProgressBar(B*2)
progress = 0
bboxes = []
for img in tqdm(images_np, total=len(images_np), desc="Detecting bboxes"):
bboxes.append(detector(
cv2.resize(img, (640, 640)).transpose(2, 0, 1)[None],
shape
)[0][0]["bbox"])
progress += 1
if progress % 10 == 0:
comfy_pbar.update_absolute(progress)
detector.cleanup()
kp2ds = []
for img, bbox in tqdm(zip(images_np, bboxes), total=len(images_np), desc="Extracting keypoints"):
if bbox is None or bbox[-1] <= 0 or (bbox[2] - bbox[0]) < 10 or (bbox[3] - bbox[1]) < 10:
bbox = np.array([0, 0, img.shape[1], img.shape[0]])
bbox_xywh = bbox
center, scale = bbox_from_detector(bbox_xywh, input_resolution, rescale=rescale)
img = crop(img, center, scale, (input_resolution[0], input_resolution[1]))[0]
img_norm = (img - IMG_NORM_MEAN) / IMG_NORM_STD
img_norm = img_norm.transpose(2, 0, 1).astype(np.float32)
keypoints = pose_model(img_norm[None], np.array(center)[None], np.array(scale)[None])
kp2ds.append(keypoints)
progress += 1
if progress % 10 == 0:
comfy_pbar.update_absolute(progress)
pose_model.cleanup()
kp2ds = np.concatenate(kp2ds, 0)
pose_metas = load_pose_metas_from_kp2ds_seq(kp2ds, width=W, height=H)
dwposes = [aaposemeta_to_dwpose_scail(meta) for meta in pose_metas]
return (dwposes,)
class RenderNLFPoses:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"nlf_poses": ("NLFPRED", {"tooltip": "Input poses for the model"}),
"width": ("INT", {"default": 512}),
"height": ("INT", {"default": 512}),
},
"optional": {
"dw_poses": ("DWPOSES", {"default": None, "tooltip": "Optional DW pose model for 2D drawing"}),
"ref_dw_pose": ("DWPOSES", {"default": None, "tooltip": "Optional reference DW pose model for alignment"}),
}
}
RETURN_TYPES = ("IMAGE", "MASK",)
RETURN_NAMES = ("image", "mask",)
FUNCTION = "predict"
CATEGORY = "WanVideoWrapper"
def predict(self, nlf_poses, width, height, dw_poses=None, ref_dw_pose=None):
from .NLFPoseExtract.nlf_render import render_nlf_as_images, shift_dwpose_according_to_nlf, process_data_to_COCO_format, intrinsic_matrix_from_field_of_view
from .NLFPoseExtract.align3d import solve_new_camera_params_central, solve_new_camera_params_down
if isinstance(nlf_poses, dict):
pose_input = nlf_poses['joints3d_nonparam'][0] if 'joints3d_nonparam' in nlf_poses else nlf_poses
else:
pose_input = nlf_poses
dw_pose_input = copy.deepcopy(dw_poses)
ori_camera_pose = intrinsic_matrix_from_field_of_view([height, width])
ori_focal = ori_camera_pose[0, 0]
if ref_dw_pose is not None:
ref_dw_pose_input = copy.deepcopy(ref_dw_pose)
pose_3d_first_driving_frame = pose_input[0][0].cpu().numpy()
pose_3d_coco_first_driving_frame = process_data_to_COCO_format(pose_3d_first_driving_frame)
poses_2d_ref = ref_dw_pose_input[0]['bodies']['candidate'][0][:14]
poses_2d_ref[:, 0] = poses_2d_ref[:, 0] * width
poses_2d_ref[:, 1] = poses_2d_ref[:, 1] * height
poses_2d_subset = ref_dw_pose[0]['bodies']['subset'][0][:14]
pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[:14]
valid_indices, valid_upper_indices, valid_lower_indices = [], [], []
upper_body_indices = [0, 2, 3, 5, 6]
lower_body_indices = [9, 10, 12, 13]
for i in range(len(poses_2d_subset)):
if poses_2d_subset[i] != -1.0 and np.sum(pose_3d_coco_first_driving_frame[i]) != 0:
if i in upper_body_indices:
valid_upper_indices.append(i)
if i in lower_body_indices:
valid_lower_indices.append(i)
valid_indices = [1] + valid_lower_indices if len(valid_upper_indices) < 4 else [1] + valid_lower_indices + valid_upper_indices # align body or only lower body
pose_2d_ref = poses_2d_ref[valid_indices]
pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[valid_indices]
if len(valid_lower_indices) >= 4:
new_camera_intrinsics, scale_m = solve_new_camera_params_down(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
else:
new_camera_intrinsics, scale_m = solve_new_camera_params_central(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
scale_face = scale_faces(list(dw_pose_input), list(ref_dw_pose)) # poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2
logging.info(f"Scale - m: {scale_m}, face: {scale_face}")
shift_dwpose_according_to_nlf(pose_input, dw_pose_input, ori_camera_pose, new_camera_intrinsics, height, width)
frames_np = render_nlf_as_images(pose_input, dw_pose_input, height, width, len(pose_input), intrinsic_matrix=new_camera_intrinsics)
else:
frames_np = render_nlf_as_images(pose_input, dw_pose_input, height, width, len(pose_input), intrinsic_matrix=ori_camera_pose)
frames_tensor = torch.from_numpy(np.stack(frames_np, axis=0)).contiguous() / 255.0
frames_tensor, mask = frames_tensor[..., :3], frames_tensor[..., -1] > 0.5
return (frames_tensor.cpu().float(), mask.cpu().float())
NODE_CLASS_MAPPINGS = {
"PoseDetectionVitPoseToDWPose": PoseDetectionVitPoseToDWPose,
"RenderNLFPoses": RenderNLFPoses,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"PoseDetectionVitPoseToDWPose": "Pose Detection VitPose to DWPose",
"RenderNLFPoses": "Render NLF Poses",
}
+221
View File
@@ -0,0 +1,221 @@
import numpy as np
def convert_3dpose_to_2dpose_body(body_keypoints, face_keypoints):
"""
Map 20-point 3D coordinates to 18-point 2D coordinates.
:param poses: Input list of 20 coordinates, each point as [x, y, z]
:return: Mapped list of 18 coordinates, each point as [x, y]
"""
# Mapping relationship: index positions
body_mapping = {
0: 1, 1: 2, 2: 3, 3: 4, 4: 5, 5: 6, 6: 7, 7: 8, 8: 8, 9: 9, 10: 10, 11: 23, 13: 22, 12: 21,
14: 11, 15: 12, 16: 13, 17: 20, 18: 18, 19: 19
}
face_mapping = {
1: 16, 8: 14, 4: 0, 7: 15, 0: 17
}
# Initialize 18-point coordinate list, default value is [-1, -1]
result = [[-1, -1] for _ in range(24)]
# Traverse the mapping relationship, map the corresponding 20-point coordinates to 18-point coordinates
for src_idx, dst_idx in body_mapping.items():
if src_idx < len(body_keypoints): # 确保索引不越界
result[dst_idx] = [body_keypoints[src_idx][1],body_keypoints[src_idx][0]] # Extract x, y coordinates
for src_idx, dst_idx in face_mapping.items():
if src_idx < len(face_keypoints):
result[dst_idx] = [face_keypoints[src_idx][1], face_keypoints[src_idx][0]]
return result
def convert_3dpose_to_2dpose_hand(left_hand_keypoints, right_hand_keypoints, body_keypoints):
"""
Map 20-point 3D coordinates to 18-point 2D coordinates.
:param poses: Input list of 20 coordinates, each point as [x, y, z]
:return: Mapped list of 18 coordinates, each point as [x, y]
"""
# Mapping relationship: index positions
hand_mapping = {
0: 1, 1: 2, 2: 3, 3: 4, 4: 5, 5: 6, 6: 7, 7: 8, 8: 9, 9: 10,
10: 11, 11: 12, 12: 13, 13: 14, 14: 15, 15: 16, 16: 17, 17: 18,
18: 19, 19: 20
}
body_mapping_left = {3: 0}
body_mapping_right = {6: 0}
# Initialize 18-point coordinate list, default value is [-1, -1]
left_result = [[-1, -1] for _ in range(21)]
right_result = [[-1, -1] for _ in range(21)]
# Traverse the mapping relationship, map the corresponding 20-point coordinates to 18-point coordinates
for src_idx, dst_idx in hand_mapping.items():
if src_idx < len(left_hand_keypoints): # 确保索引不越界
left_result[dst_idx] = [left_hand_keypoints[src_idx][1], left_hand_keypoints[src_idx][0]] # Extract x, y coordinates
right_result[dst_idx] = [right_hand_keypoints[src_idx][1], right_hand_keypoints[src_idx][0]]
for src_idx, dst_idx in body_mapping_left.items():
if src_idx < len(body_keypoints):
left_result[dst_idx] = [body_keypoints[src_idx][1], body_keypoints[src_idx][0]]
for src_idx, dst_idx in body_mapping_right.items():
if src_idx < len(body_keypoints):
right_result[dst_idx] = [body_keypoints[src_idx][1], body_keypoints[src_idx][0]]
return [left_result, right_result]
def convert_3dpose_to_2dpose_face(face_keypoints):
# Set [-1, -1] for indices 0, 1, 4, 5, 6, 7, 8, otherwise extract [y, x]
result = [[-1, -1] if i in [0, 1, 4, 5, 6, 7, 8] else [pt[1], pt[0]] for i, pt in enumerate(face_keypoints)]
return result
def correct_lift_end_kpt_by_phmr(start, end, dwpose_kpts, lift_start, lift_end, phmr_start, phmr_end):
'''
Check if the other end meets the requirements. If so, return the result after lift, otherwise return the phmr result.
'''
if dwpose_kpts[start][0] == -1:
return
lift_vec = np.array(lift_end) - np.array(lift_start)
phmr_vec = np.array(phmr_end) - np.array(phmr_start)
start_distance = np.linalg.norm(np.array(lift_start) - np.array(phmr_start))
end_distance = np.linalg.norm(np.array(lift_end) - np.array(phmr_end))
lift_vec_len = np.linalg.norm(lift_vec)
phmr_vec_len = np.linalg.norm(phmr_vec)
if start_distance + end_distance > phmr_vec_len:
dwpose_kpts[end] = [-1, -1]
theta = np.arccos(np.dot(lift_vec, phmr_vec) / (lift_vec_len * phmr_vec_len))
if lift_vec_len > phmr_vec_len * 1.65 or lift_vec_len < phmr_vec_len * 0.4 or theta > np.pi / 4:
dwpose_kpts[end] = [-1, -1]
return
def mix_3d_poses(poses_dwpose, poses_3dpose):
'''
Combine two types of poses: use the body from 3dPose, and the face and hand from DWPose.
'''
poses = []
for pose_dwpose, pose_3dpose in zip(poses_dwpose, poses_3dpose):
pose = {
"bodies": {
"candidate": pose_3dpose["bodies"]["candidate"],
"subset": pose_dwpose["bodies"]["subset"]
},
"faces": pose_dwpose["faces"],
"hands": pose_dwpose["hands"]
}
poses.append(pose)
return poses
def correct_hand_from_3d(hand_keypoints_dwpose, hand_keypoints_3dpose):
'''
If the hand keypoints of dwpose and 3dpose differ too much, remove the farthest end.
'''
edges_palm = [
[1, 2], [2, 3], [3, 4],
[5, 6], [6, 7], [7, 8],
[9, 10], [10, 11], [11, 12],
[13, 14], [14, 15], [15, 16],
[17, 18], [18, 19], [19, 20],
]
edges_finger = [[0, 1], [0, 5], [0, 9], [0, 13], [0, 17]]
max_length_palm = 0
max_length_finger = 0
for edge in edges_palm:
limb_length_3dpose = np.linalg.norm(np.array(hand_keypoints_3dpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[1]]))
if limb_length_3dpose > max_length_palm:
max_length_palm = limb_length_3dpose
for edge in edges_finger:
limb_length_3dpose = np.linalg.norm(np.array(hand_keypoints_3dpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[1]]))
if limb_length_3dpose > max_length_finger:
max_length_finger = limb_length_3dpose
for edge in edges_palm:
limb_length_dwpose = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_dwpose[edge[1]]))
if limb_length_dwpose > max_length_palm * 1.5:
if -1 in hand_keypoints_dwpose[edge[0]] or -1 in hand_keypoints_dwpose[edge[1]] or -1 in hand_keypoints_3dpose[edge[0]] or -1 in hand_keypoints_3dpose[edge[1]]:
continue
distance_point_0 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[0]]))
distance_point_1 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[1]]) - np.array(hand_keypoints_3dpose[edge[1]]))
if distance_point_0 > distance_point_1:
hand_keypoints_dwpose[edge[1]] = [-1, -1]
else:
hand_keypoints_dwpose[edge[0]] = [-1, -1]
for edge in edges_finger:
limb_length_dwpose = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_dwpose[edge[1]]))
if limb_length_dwpose > max_length_finger * 1.5:
if -1 in hand_keypoints_dwpose[edge[0]] or -1 in hand_keypoints_dwpose[edge[1]] or -1 in hand_keypoints_3dpose[edge[0]] or -1 in hand_keypoints_3dpose[edge[1]]:
continue
distance_point_0 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[0]]))
distance_point_1 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[1]]) - np.array(hand_keypoints_3dpose[edge[1]]))
if distance_point_0 > distance_point_1:
hand_keypoints_dwpose[edge[1]] = [-1, -1]
else:
hand_keypoints_dwpose[edge[0]] = [-1, -1]
return hand_keypoints_dwpose
def correct_body_from_3d(body_keypoints_dwpose, body_keypoints_3dpose, subset_dwpose, subset_3dpose):
'''
If the bone length of dwpose and 3dpose differ too much, remove the farthest end.
'''
limbSeq = [
[2, 3],
[2, 6],
[3, 4],
[4, 5],
[6, 7],
[7, 8],
[2, 9],
[9, 10],
[10, 11],
[2, 12],
[12, 13],
[13, 14],
[2, 1],
[1, 15],
[15, 17],
[1, 16],
[16, 18],
[3, 17],
[6, 18],
]
for ori_limb in limbSeq:
limb = [ori_limb[0] - 1, ori_limb[1] - 1]
limb_length_dwpose = np.linalg.norm(np.array(body_keypoints_dwpose[limb[0]]) - np.array(body_keypoints_dwpose[limb[1]]))
limb_length_3dpose = np.linalg.norm(np.array(body_keypoints_3dpose[limb[0]]) - np.array(body_keypoints_3dpose[limb[1]]))
if subset_dwpose[0][limb[0]] == -1 or subset_dwpose[0][limb[1]] == -1 or subset_3dpose[0][limb[0]] == -1 or subset_3dpose[0][limb[1]] == -1:
continue
if limb_length_dwpose > limb_length_3dpose * 2:
# Determine the farther end
distance_point_0 = np.linalg.norm(np.array(body_keypoints_dwpose[limb[0]]) - np.array(body_keypoints_3dpose[limb[0]]))
distance_point_1 = np.linalg.norm(np.array(body_keypoints_dwpose[limb[1]]) - np.array(body_keypoints_3dpose[limb[1]]))
if distance_point_0 > distance_point_1:
if limb[1] == 1: # core
continue
body_keypoints_dwpose[limb[1]] = [-1, -1]
subset_dwpose[0][limb[1]] = -1
else:
if limb[0] == 1: # core
continue
body_keypoints_dwpose[limb[0]] = [-1, -1]
subset_dwpose[0][limb[0]] = -1
return body_keypoints_dwpose, subset_dwpose
def correct_full_pose_from_3d(poses_dwpose, poses_3dpose):
'''
If the bone length of dwpose and 3dpose differ too much, remove the end farthest from the 3d pose.
'''
poses = []
for pose_dwpose, pose_3dpose in zip(poses_dwpose, poses_3dpose):
new_candidate, new_subset = correct_body_from_3d(pose_dwpose["bodies"]["candidate"], pose_3dpose["bodies"]["candidate"], pose_dwpose["bodies"]["subset"], pose_3dpose["bodies"]["subset"])
new_hands_0 = correct_hand_from_3d(pose_dwpose["hands"][0], pose_3dpose["hands"][0])
new_hands_1 = correct_hand_from_3d(pose_dwpose["hands"][1], pose_3dpose["hands"][1])
pose = {
"bodies": {
"candidate": new_candidate,
"subset": new_subset
},
"faces": pose_dwpose["faces"],
"hands": [new_hands_0, new_hands_1]
}
poses.append(pose)
return poses
+129
View File
@@ -0,0 +1,129 @@
import cv2
import numpy as np
from PIL import Image
import os
from .draw_utils import draw_bodypose, draw_bodypose_with_feet, draw_handpose_lr, draw_handpose, draw_facepose, draw_bodypose_augmentation
def draw_pose(pose, H, W, show_feet=False, show_body=True, show_hand=True, show_face=True, show_cheek=False, dw_bgr=False, dw_hand=False, aug_body_draw=False, optimized_face=False):
final_canvas = np.zeros(shape=(H, W, 3), dtype=np.uint8)
for i in range(len(pose["bodies"]["candidate"])):
canvas = np.zeros(shape=(H, W, 3), dtype=np.uint8)
bodies = pose["bodies"]
faces = pose["faces"][i:i+1]
hands = pose["hands"][2*i:2*i+2]
candidate = bodies["candidate"][i]
subset = bodies["subset"][i:i+1]
if show_body:
if len(subset[0]) <= 18 or show_feet == False:
if aug_body_draw:
raise NotImplementedError("aug_body_draw is not implemented yet")
else:
canvas = draw_bodypose(canvas, candidate, subset)
else:
canvas = draw_bodypose_with_feet(canvas, candidate, subset)
if dw_bgr:
canvas = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB)
if show_cheek:
assert show_body == False, "show_cheek and show_body cannot be True at the same time"
canvas = draw_bodypose_augmentation(canvas, candidate, subset, drop_aug=True, shift_aug=False, all_cheek_aug=True)
if show_hand:
if not dw_hand:
canvas = draw_handpose_lr(canvas, hands)
else:
canvas = draw_handpose(canvas, hands)
if show_face:
canvas = draw_facepose(canvas, faces, optimized_face=optimized_face)
final_canvas = final_canvas + canvas
return final_canvas
def scale_image_hw_keep_size(img, scale_h, scale_w):
"""Scale the image by scale_h and scale_w respectively, keeping the output size unchanged."""
H, W = img.shape[:2]
new_H, new_W = int(H * scale_h), int(W * scale_w)
scaled = cv2.resize(img, (new_W, new_H), interpolation=cv2.INTER_LINEAR)
result = np.zeros_like(img)
# 计算在目标图上的放置范围
# --- Y方向 ---
if new_H >= H:
y_start_src = (new_H - H) // 2
y_end_src = y_start_src + H
y_start_dst = 0
y_end_dst = H
else:
y_start_src = 0
y_end_src = new_H
y_start_dst = (H - new_H) // 2
y_end_dst = y_start_dst + new_H
# --- X方向 ---
if new_W >= W:
x_start_src = (new_W - W) // 2
x_end_src = x_start_src + W
x_start_dst = 0
x_end_dst = W
else:
x_start_src = 0
x_end_src = new_W
x_start_dst = (W - new_W) // 2
x_end_dst = x_start_dst + new_W
# 将 scaled 映射到 result
result[y_start_dst:y_end_dst, x_start_dst:x_end_dst] = scaled[y_start_src:y_end_src, x_start_src:x_end_src]
return result
def draw_pose_to_canvas_np(poses, pool, H, W, reshape_scale, show_feet_flag=False, show_body_flag=True, show_hand_flag=True, show_face_flag=True, show_cheek_flag=False, dw_bgr=False, dw_hand=False, aug_body_draw=False):
canvas_np_lst = []
for pose in poses:
if reshape_scale > 0:
pool.apply_random_reshapes(pose)
canvas = draw_pose(pose, H, W, show_feet_flag, show_body_flag, show_hand_flag, show_face_flag, show_cheek_flag, dw_bgr, dw_hand, aug_body_draw, optimized_face=True)
canvas_np_lst.append(canvas)
return canvas_np_lst
def draw_pose_to_canvas(poses, pool, H, W, reshape_scale, points_only_flag, show_feet_flag, show_body_flag=True, show_hand_flag=True, show_face_flag=True, show_cheek_flag=False, dw_bgr=False, dw_hand=False, aug_body_draw=False):
canvas_lst = []
for pose in poses:
if reshape_scale > 0:
pool.apply_random_reshapes(pose)
canvas = draw_pose(pose, H, W, show_feet_flag, show_body_flag, show_hand_flag, show_face_flag, show_cheek_flag, dw_bgr, dw_hand, aug_body_draw, optimized_face=False)
canvas_img = Image.fromarray(canvas)
canvas_lst.append(canvas_img)
return canvas_lst
def get_mp4_filenames_from_directory(dwpose_keypoints_dir):
mp4_filenames_dwpose = []
# Get all available mp4 files by intersecting keypoints and mp4
if dwpose_keypoints_dir:
for root, dirs, files in os.walk(dwpose_keypoints_dir):
for file in files:
if file.lower().endswith('.pt'): # Only look for .mp4 files
mp4_filenames_dwpose.append(file.replace(".pt", ".mp4")) # Get absolute path
return mp4_filenames_dwpose
def project_dwpose_to_3d(dwpose_keypoint, original_threed_keypoint, focal, princpt, H, W):
# Camera intrinsic parameters
# fx, fy = focal, focal
fx, fy = focal
cx, cy = princpt
# 2D keypoint coordinates
x_2d, y_2d = dwpose_keypoint[0] * W, dwpose_keypoint[1] * H
# Original 3D point (in camera coordinate system)
ori_x, ori_y, ori_z = original_threed_keypoint
# Use the new 2D point and original depth to compute the new 3D point by back-projection
# Formula: x = (u - cx) * z / fx
new_x = (x_2d - cx) * ori_z / fx
new_y = (y_2d - cy) * ori_z / fy
new_z = ori_z # Keep the depth unchanged
return [new_x, new_y, new_z]
+658
View File
@@ -0,0 +1,658 @@
# https://github.com/IDEA-Research/DWPose
import math
import numpy as np
import matplotlib
import cv2
import random
eps = 0.01
def smart_resize(x, s):
Ht, Wt = s
if x.ndim == 2:
Ho, Wo = x.shape
Co = 1
else:
Ho, Wo, Co = x.shape
if Co == 3 or Co == 1:
k = float(Ht + Wt) / float(Ho + Wo)
return cv2.resize(
x,
(int(Wt), int(Ht)),
interpolation=cv2.INTER_AREA if k < 1 else cv2.INTER_LANCZOS4,
)
else:
return np.stack([smart_resize(x[:, :, i], s) for i in range(Co)], axis=2)
def smart_resize_k(x, fx, fy):
if x.ndim == 2:
Ho, Wo = x.shape
Co = 1
else:
Ho, Wo, Co = x.shape
Ht, Wt = Ho * fy, Wo * fx
if Co == 3 or Co == 1:
k = float(Ht + Wt) / float(Ho + Wo)
return cv2.resize(
x,
(int(Wt), int(Ht)),
interpolation=cv2.INTER_AREA if k < 1 else cv2.INTER_LANCZOS4,
)
else:
return np.stack([smart_resize_k(x[:, :, i], fx, fy) for i in range(Co)], axis=2)
def padRightDownCorner(img, stride, padValue):
h = img.shape[0]
w = img.shape[1]
pad = 4 * [None]
pad[0] = 0 # up
pad[1] = 0 # left
pad[2] = 0 if (h % stride == 0) else stride - (h % stride) # down
pad[3] = 0 if (w % stride == 0) else stride - (w % stride) # right
img_padded = img
pad_up = np.tile(img_padded[0:1, :, :] * 0 + padValue, (pad[0], 1, 1))
img_padded = np.concatenate((pad_up, img_padded), axis=0)
pad_left = np.tile(img_padded[:, 0:1, :] * 0 + padValue, (1, pad[1], 1))
img_padded = np.concatenate((pad_left, img_padded), axis=1)
pad_down = np.tile(img_padded[-2:-1, :, :] * 0 + padValue, (pad[2], 1, 1))
img_padded = np.concatenate((img_padded, pad_down), axis=0)
pad_right = np.tile(img_padded[:, -2:-1, :] * 0 + padValue, (1, pad[3], 1))
img_padded = np.concatenate((img_padded, pad_right), axis=1)
return img_padded, pad
def transfer(model, model_weights):
transfered_model_weights = {}
for weights_name in model.state_dict().keys():
transfered_model_weights[weights_name] = model_weights[
".".join(weights_name.split(".")[1:])
]
return transfered_model_weights
def draw_bodypose_with_feet(canvas, candidate, subset):
H, W, C = canvas.shape
candidate = np.array(candidate)
subset = np.array(subset)
stickwidth = 4
# 原始18个关节点的连接顺序(和 OpenPose 的 COCO 模型一致)
limbSeq = [
[2, 3],
[2, 6],
[3, 4],
[4, 5],
[6, 7],
[7, 8],
[2, 9],
[9, 10],
[10, 11],
[2, 12],
[12, 13],
[13, 14],
[2, 1],
[1, 15],
[15, 17],
[1, 16],
[16, 18],
[3, 17],
[6, 18],
]
# 添加脚部连接线:10->18, 10->19, 10->20;13->21, 13->22, 13->23
foot_limbSeq = [
[14, 19],
[14, 20],
[14, 21],
[11, 22],
[11, 23],
[11, 24],
]
# 生成颜色(原始18条颜色 + 6条新颜色)
colors = [
[255, 0, 0],
[255, 85, 0],
[255, 170, 0],
[255, 255, 0],
[170, 255, 0],
[85, 255, 0],
[0, 255, 0],
[0, 255, 85],
[0, 255, 170],
[0, 255, 255],
[0, 170, 255],
[0, 85, 255],
[0, 0, 255],
[85, 0, 255],
[170, 0, 255],
[255, 0, 255],
[255, 0, 170],
[255, 0, 85],
]
colors_feet = [
[100, 0, 215], [80, 0, 235], [60, 0, 255],
[0, 235, 150], [0, 215, 170], [0, 195, 190],
]
colors = colors + colors_feet
for i in range(17):
for n in range(len(subset)):
index = subset[n][np.array(limbSeq[i]) - 1]
if -1 in index:
continue
Y = candidate[index.astype(int), 0] * float(W)
X = candidate[index.astype(int), 1] * float(H)
mX = np.mean(X)
mY = np.mean(Y)
length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
polygon = cv2.ellipse2Poly(
(int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
)
cv2.fillConvexPoly(canvas, polygon, colors[i])
for i in range(6):
for n in range(len(subset)):
index = subset[n][np.array(foot_limbSeq[i]) - 1]
if -1 in index:
continue
Y = candidate[index.astype(int), 0] * float(W)
X = candidate[index.astype(int), 1] * float(H)
mX = np.mean(X)
mY = np.mean(Y)
length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
polygon = cv2.ellipse2Poly(
(int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
)
cv2.fillConvexPoly(canvas, polygon, colors_feet[i])
canvas = (canvas * 0.6).astype(np.uint8)
# 画关键点
for i in range(24):
for n in range(len(subset)):
index = int(subset[n][i])
if index == -1:
continue
x, y = candidate[index][0:2]
x = int(x * W)
y = int(y * H)
cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1)
return canvas
def draw_bodypose_augmentation(canvas, candidate, subset, drop_aug=True, shift_aug=False, all_cheek_aug=False):
H, W, C = canvas.shape
candidate = np.array(candidate)
subset = np.array(subset)
stickwidth = 4
limbSeq = [
[2, 3], # 1->2 左肩 0
[2, 6], # 1->5 右肩 1
[3, 4], # 2->3 左臂 2
[4, 5], # 3->4 左肘 3
[6, 7], # 5->6 右臂 4
[7, 8], # 6->7 右肘 5
[2, 9], # 6
[9, 10], # 7
[10, 11], # 8
[2, 12], # 9
[12, 13], # 10
[13, 14], # 11
[2, 1], # 12
[1, 15], # 13 cheek
[15, 17], # 14 cheek
[1, 16], # 15 cheek
[16, 18], # 16 cheek
[3, 17],
[6, 18],
]
colors = [
[255, 0, 0],
[255, 85, 0],
[255, 170, 0],
[255, 255, 0],
[170, 255, 0],
[85, 255, 0],
[0, 255, 0],
[0, 255, 85],
[0, 255, 170],
[0, 255, 255],
[0, 170, 255],
[0, 85, 255],
[0, 0, 255],
[85, 0, 255],
[170, 0, 255],
[255, 0, 255],
[255, 0, 170],
[255, 0, 85],
]
# 随机选0-2根骨骼进行丢弃
if drop_aug:
arr_drop = list(range(17))
k_drop = random.choices([0, 1, 2], weights=[0.5, 0.3, 0.2])[0]
drop_indices = random.sample(arr_drop, k_drop)
else:
drop_indices = []
if shift_aug:
shift_indices = random.sample(list(range(17)), 2)
else:
shift_indices = []
if all_cheek_aug:
drop_indices = list(range(13)) # 0-12对应的骨骼都扔掉
for i in range(17):
for n in range(len(subset)):
index = subset[n][np.array(limbSeq[i]) - 1]
if -1 in index:
continue
Y = candidate[index.astype(int), 0] * float(W)
X = candidate[index.astype(int), 1] * float(H)
if i in drop_indices:
continue
mX = np.mean(X) # 计算两个关节点之间的中点
mY = np.mean(Y)
length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
if i in shift_indices:
mX = mX + random.uniform(-length/4, length/4)
mY = mY + random.uniform(-length/4, length/4)
angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
polygon = cv2.ellipse2Poly(
(int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
)
cv2.fillConvexPoly(canvas, polygon, colors[i])
canvas = (canvas * 0.6).astype(np.uint8)
for i in range(18):
if all_cheek_aug:
if not i in [0, 14, 15, 16, 17]:
continue
for n in range(len(subset)):
index = int(subset[n][i])
if index == -1:
continue
x, y = candidate[index][0:2]
x = int(x * W)
y = int(y * H)
cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1)
return canvas
def draw_bodypose(canvas, candidate, subset):
H, W, C = canvas.shape
candidate = np.array(candidate)
subset = np.array(subset)
stickwidth = 4
limbSeq = [
[2, 3],
[2, 6],
[3, 4],
[4, 5],
[6, 7],
[7, 8],
[2, 9],
[9, 10],
[10, 11],
[2, 12],
[12, 13],
[13, 14],
[2, 1],
[1, 15],
[15, 17],
[1, 16],
[16, 18],
[3, 17],
[6, 18],
]
colors = [
[255, 0, 0],
[255, 85, 0],
[255, 170, 0],
[255, 255, 0],
[170, 255, 0],
[85, 255, 0],
[0, 255, 0],
[0, 255, 85],
[0, 255, 170],
[0, 255, 255],
[0, 170, 255],
[0, 85, 255],
[0, 0, 255],
[85, 0, 255],
[170, 0, 255],
[255, 0, 255],
[255, 0, 170],
[255, 0, 85],
]
for i in range(17):
for n in range(len(subset)):
index = subset[n][np.array(limbSeq[i]) - 1]
if -1 in index:
continue
Y = candidate[index.astype(int), 0] * float(W)
X = candidate[index.astype(int), 1] * float(H)
mX = np.mean(X)
mY = np.mean(Y)
length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
polygon = cv2.ellipse2Poly(
(int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
)
cv2.fillConvexPoly(canvas, polygon, colors[i])
canvas = (canvas * 0.6).astype(np.uint8)
for i in range(18):
for n in range(len(subset)):
index = int(subset[n][i])
if index == -1:
continue
x, y = candidate[index][0:2]
x = int(x * W)
y = int(y * H)
cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1)
return canvas
def draw_handpose_lr(canvas, all_hand_peaks):
H, W, C = canvas.shape
# 连接顺序:21个关键点的骨架连线
edges = [
[0, 1], [1, 2], [2, 3], [3, 4],
[0, 5], [5, 6], [6, 7], [7, 8],
[0, 9], [9, 10], [10, 11], [11, 12],
[0, 13], [13, 14], [14, 15], [15, 16],
[0, 17], [17, 18], [18, 19], [19, 20],
]
all_num_hands = len(all_hand_peaks)
for peaks_idx, peaks in enumerate(all_hand_peaks):
left_or_right = not (peaks_idx >= all_num_hands / 2)
base_hue = 0 if left_or_right == 0 else 0.3
peaks = np.array(peaks)
for ie, e in enumerate(edges):
x1, y1 = peaks[e[0]]
x2, y2 = peaks[e[1]]
x1 = int(x1 * W)
y1 = int(y1 * H)
x2 = int(x2 * W)
y2 = int(y2 * H)
if x1 > eps and y1 > eps and x2 > eps and y2 > eps:
if left_or_right == 0:
hsv_color = [ (base_hue + ie / float(len(edges)) * 0.8), 0.9, 0.9 ]
else:
hsv_color = [ (base_hue + ie / float(len(edges)) * 0.8), 0.8, 1 ]
rgb_color = matplotlib.colors.hsv_to_rgb(hsv_color) * 255
cv2.line(
canvas,
(x1, y1),
(x2, y2),
rgb_color,
thickness=2,
)
for i, keypoint in enumerate(peaks):
x, y = keypoint
x = int(x * W)
y = int(y * H)
if x > eps and y > eps:
# 关键点也用淡色标注(左手蓝、右手红)
point_color = (245, 100, 100) if left_or_right == 0 else (100, 100, 255)
cv2.circle(canvas, (x, y), 4, point_color, thickness=-1)
return canvas
def draw_handpose(canvas, all_hand_peaks):
H, W, C = canvas.shape
stickwidth_thin = min(max(int(min(H, W) / 300), 1), 2)
edges = [
[0, 1],
[1, 2],
[2, 3],
[3, 4],
[0, 5],
[5, 6],
[6, 7],
[7, 8],
[0, 9],
[9, 10],
[10, 11],
[11, 12],
[0, 13],
[13, 14],
[14, 15],
[15, 16],
[0, 17],
[17, 18],
[18, 19],
[19, 20],
]
for peaks in all_hand_peaks:
peaks = np.array(peaks)
for ie, e in enumerate(edges):
x1, y1 = peaks[e[0]]
x2, y2 = peaks[e[1]]
x1 = int(x1 * W)
y1 = int(y1 * H)
x2 = int(x2 * W)
y2 = int(y2 * H)
if x1 > eps and y1 > eps and x2 > eps and y2 > eps:
cv2.line(
canvas,
(x1, y1),
(x2, y2),
matplotlib.colors.hsv_to_rgb([ie / float(len(edges)), 1.0, 1.0])
* 255,
thickness=stickwidth_thin,
)
for i, keyponit in enumerate(peaks):
x, y = keyponit
x = int(x * W)
y = int(y * H)
if x > eps and y > eps:
cv2.circle(canvas, (x, y), stickwidth_thin, (0, 0, 255), thickness=-1)
return canvas
def draw_facepose(canvas, all_lmks, optimized_face=True):
H, W, C = canvas.shape
stickwidth = min(max(int(min(H, W) / 200), 1), 3)
stickwidth_thin = min(max(int(min(H, W) / 300), 1), 2)
for lmks in all_lmks:
lmks = np.array(lmks)
for lmk_idx, lmk in enumerate(lmks):
x, y = lmk
x = int(x * W)
y = int(y * H)
if x > eps and y > eps:
if optimized_face:
if lmk_idx in list(range(17, 27)) + list(range(36, 70)):
cv2.circle(canvas, (x, y), stickwidth_thin, (255, 255, 255), thickness=-1)
else:
cv2.circle(canvas, (x, y), stickwidth, (255, 255, 255), thickness=-1)
return canvas
# detect hand according to body pose keypoints
# please refer to https://github.com/CMU-Perceptual-Computing-Lab/openpose/blob/master/src/openpose/hand/handDetector.cpp
def handDetect(candidate, subset, oriImg):
# right hand: wrist 4, elbow 3, shoulder 2
# left hand: wrist 7, elbow 6, shoulder 5
ratioWristElbow = 0.33
detect_result = []
image_height, image_width = oriImg.shape[0:2]
for person in subset.astype(int):
# if any of three not detected
has_left = np.sum(person[[5, 6, 7]] == -1) == 0
has_right = np.sum(person[[2, 3, 4]] == -1) == 0
if not (has_left or has_right):
continue
hands = []
# left hand
if has_left:
left_shoulder_index, left_elbow_index, left_wrist_index = person[[5, 6, 7]]
x1, y1 = candidate[left_shoulder_index][:2]
x2, y2 = candidate[left_elbow_index][:2]
x3, y3 = candidate[left_wrist_index][:2]
hands.append([x1, y1, x2, y2, x3, y3, True])
# right hand
if has_right:
right_shoulder_index, right_elbow_index, right_wrist_index = person[
[2, 3, 4]
]
x1, y1 = candidate[right_shoulder_index][:2]
x2, y2 = candidate[right_elbow_index][:2]
x3, y3 = candidate[right_wrist_index][:2]
hands.append([x1, y1, x2, y2, x3, y3, False])
for x1, y1, x2, y2, x3, y3, is_left in hands:
# pos_hand = pos_wrist + ratio * (pos_wrist - pos_elbox) = (1 + ratio) * pos_wrist - ratio * pos_elbox
# handRectangle.x = posePtr[wrist*3] + ratioWristElbow * (posePtr[wrist*3] - posePtr[elbow*3]);
# handRectangle.y = posePtr[wrist*3+1] + ratioWristElbow * (posePtr[wrist*3+1] - posePtr[elbow*3+1]);
# const auto distanceWristElbow = getDistance(poseKeypoints, person, wrist, elbow);
# const auto distanceElbowShoulder = getDistance(poseKeypoints, person, elbow, shoulder);
# handRectangle.width = 1.5f * fastMax(distanceWristElbow, 0.9f * distanceElbowShoulder);
x = x3 + ratioWristElbow * (x3 - x2)
y = y3 + ratioWristElbow * (y3 - y2)
distanceWristElbow = math.sqrt((x3 - x2) ** 2 + (y3 - y2) ** 2)
distanceElbowShoulder = math.sqrt((x2 - x1) ** 2 + (y2 - y1) ** 2)
width = 1.5 * max(distanceWristElbow, 0.9 * distanceElbowShoulder)
# x-y refers to the center --> offset to topLeft point
# handRectangle.x -= handRectangle.width / 2.f;
# handRectangle.y -= handRectangle.height / 2.f;
x -= width / 2
y -= width / 2 # width = height
# overflow the image
if x < 0:
x = 0
if y < 0:
y = 0
width1 = width
width2 = width
if x + width > image_width:
width1 = image_width - x
if y + width > image_height:
width2 = image_height - y
width = min(width1, width2)
# the max hand box value is 20 pixels
if width >= 20:
detect_result.append([int(x), int(y), int(width), is_left])
"""
return value: [[x, y, w, True if left hand else False]].
width=height since the network require squared input.
x, y is the coordinate of top left
"""
return detect_result
# Written by Lvmin
def faceDetect(candidate, subset, oriImg):
# left right eye ear 14 15 16 17
detect_result = []
image_height, image_width = oriImg.shape[0:2]
for person in subset.astype(int):
has_head = person[0] > -1
if not has_head:
continue
has_left_eye = person[14] > -1
has_right_eye = person[15] > -1
has_left_ear = person[16] > -1
has_right_ear = person[17] > -1
if not (has_left_eye or has_right_eye or has_left_ear or has_right_ear):
continue
head, left_eye, right_eye, left_ear, right_ear = person[[0, 14, 15, 16, 17]]
width = 0.0
x0, y0 = candidate[head][:2]
if has_left_eye:
x1, y1 = candidate[left_eye][:2]
d = max(abs(x0 - x1), abs(y0 - y1))
width = max(width, d * 3.0)
if has_right_eye:
x1, y1 = candidate[right_eye][:2]
d = max(abs(x0 - x1), abs(y0 - y1))
width = max(width, d * 3.0)
if has_left_ear:
x1, y1 = candidate[left_ear][:2]
d = max(abs(x0 - x1), abs(y0 - y1))
width = max(width, d * 1.5)
if has_right_ear:
x1, y1 = candidate[right_ear][:2]
d = max(abs(x0 - x1), abs(y0 - y1))
width = max(width, d * 1.5)
x, y = x0, y0
x -= width
y -= width
if x < 0:
x = 0
if y < 0:
y = 0
width1 = width * 2
width2 = width * 2
if x + width > image_width:
width1 = image_width - x
if y + width > image_height:
width2 = image_height - y
width = min(width1, width2)
if width >= 20:
detect_result.append([int(x), int(y), int(width)])
return detect_result
# get max index of 2d array
def npmax(array):
arrayindex = array.argmax(1)
arrayvalue = array.max(1)
i = arrayvalue.argmax()
j = arrayindex[i]
return i, j
+15
View File
@@ -0,0 +1,15 @@
[project]
name = "ComfyUI-SCAIL-Pose"
description = "ComfyUI nodes for SCAIL input processing"
version = "1.0.0"
license = {file = "LICENSE"}
dependencies = ["taichi", "pyrender", "trimesh", "opencv-python", "matplotlib", "pillow"]
[project.urls]
Repository = "https://github.com/kijai/ComfyUI-SCAIL-Pose"
# Used by Comfy Registry https://comfyregistry.org
[tool.comfy]
PublisherId = "kijai"
DisplayName = "ComfyUI-SCAIL-Pose"
Icon = ""
+12
View File
@@ -0,0 +1,12 @@
# ComfyUI nodes for SCAIL-pose processing
The code is cleaned, simplified version of: https://github.com/zai-org/SCAIL-Pose
For face and hands, instead of DWPose this uses Vitpose and it's outputs converted into DWpose format for the optional alignment
VitPose detector is available in these nodes: https://github.com/kijai/ComfyUI-WanAnimatePreprocess
NLF model loader is already included in WanVideoWrapper
Reason this is separate repository is the additional requirements of `taichi` and `pyrender`
+93
View File
@@ -0,0 +1,93 @@
import numpy as np
import cv2
from PIL import Image
import pyrender
import trimesh
def render_colored_cylinders(cylinder_specs, focal, princpt, image_size=(1280, 1280), img=None):
H, W = image_size
if isinstance(focal, float) or isinstance(focal, int):
fx, fy = focal, focal
else:
fx, fy = focal[0], focal[1]
cx, cy = princpt
# Initialize scene
scene = pyrender.Scene(bg_color=[0, 0, 0, 0], ambient_light=[0.1, 0.1, 0.1])
# Set up camera
camera = pyrender.IntrinsicsCamera(fx=fx, fy=fy, cx=cx, cy=cy, znear=0.5, zfar=10000)
pyrender2opencv = np.array([[1.0, 0, 0, 0],
[0, -1, 0, 0],
[0, 0, -1, 0],
[0, 0, 0, 1]])
cam_pose = pyrender2opencv @ np.eye(4)
scene.add(camera, pose=cam_pose)
# Add light source
light = pyrender.DirectionalLight(color=np.ones(3), intensity=3.0)
scene.add(light, pose=cam_pose)
points_to_draw = []
for start, end, color in cylinder_specs:
start = np.array(start)
end = np.array(end)
vec = end - start
height = np.linalg.norm(vec)
if height == 0:
continue
tm = trimesh.creation.cylinder(radius=12, height=height, sections=16)
# Rotate to align with z-axis
z_axis = np.array([0, 0, 1])
axis = np.cross(z_axis, vec)
if np.linalg.norm(axis) > 1e-6:
axis = axis / np.linalg.norm(axis)
angle = np.arccos(np.dot(z_axis, vec) / height)
rot = trimesh.transformations.rotation_matrix(angle, axis)
tm.apply_transform(rot)
tm.apply_translation(start + vec / 2)
# Material color (supports RGBA)
rgba = np.array(color)
material = pyrender.MetallicRoughnessMaterial(
metallicFactor=0.1,
roughnessFactor=0.5,
baseColorFactor=rgba
)
mesh = pyrender.Mesh.from_trimesh(tm, material=material)
scene.add(mesh)
# Projected points for visualization, check if projection is correct
x1 = fx * (start[0] / start[2]) + cx
y1 = fy * (start[1] / start[2]) + cy
x2 = fx * (end[0] / end[2]) + cx
y2 = fy * (end[1] / end[2]) + cy
points_to_draw.append((x1, y1))
points_to_draw.append((x2, y2))
# Render
r = pyrender.OffscreenRenderer(viewport_width=W, viewport_height=H, point_size=1.0)
color, _ = r.render(scene, flags=pyrender.RenderFlags.RGBA)
# Post-processing
color = color.astype(np.float32) / 255.0
final_img = (color * 255).astype(np.uint8)
# Draw points, check if projection is correct
for (x, y) in points_to_draw:
print(f" debug point: {x}, {y}")
x_draw = int(x)
y_draw = int(y)
cv2.circle(final_img, (x_draw, y_draw), radius=4, color=(0, 255, 0), thickness=-1)
return Image.fromarray(final_img)
+207
View File
@@ -0,0 +1,207 @@
import taichi as ti
import numpy as np
import random
import math
ti.init(arch=ti.cuda)
def flatten_specs(specs_list):
"""把 specs_list 拉平为 numpy 数组 + 索引表"""
starts, ends, colors = [], [], []
frame_offset, frame_count = [], []
offset = 0
for specs in specs_list:
frame_offset.append(offset)
frame_count.append(len(specs))
for (s, e, c) in specs:
starts.append(s)
ends.append(e)
colors.append(c)
offset += len(specs)
return (
np.array(starts, dtype=np.float32),
np.array(ends, dtype=np.float32),
np.array(colors, dtype=np.float32),
np.array(frame_offset, dtype=np.int32),
np.array(frame_count, dtype=np.int32),
)
def render_whole(specs_list, H=480, W=640, fx=500, fy=500, cx=240, cy=320, radius=21.5):
img = ti.Vector.field(4, dtype=ti.f32, shape=(H, W))
starts, ends, colors, frame_offset, frame_count = flatten_specs(specs_list)
total_cyl = len(starts)
n_frames = len(specs_list)
z_min = min(starts[:, 2].min(), ends[:, 2].min())
z_max = max(starts[:, 2].max(), ends[:, 2].max())
# ========= 相机内参 =========
znear = 0.1
zfar = max(min(z_max, 25000), 10000)
C = ti.Vector([0.0, 0.0, 0.0]) # 相机中心
light_dir = ti.Vector([0.0, 0.0, 1.0])
c_start = ti.Vector.field(3, dtype=ti.f32, shape=total_cyl)
c_end = ti.Vector.field(3, dtype=ti.f32, shape=total_cyl)
c_rgba = ti.Vector.field(4, dtype=ti.f32, shape=total_cyl)
n_cyl = ti.field(dtype=ti.i32, shape=()) # 实际数量
f_offset = ti.field(dtype=ti.i32, shape=n_frames)
f_count = ti.field(dtype=ti.i32, shape=n_frames)
frame_id = ti.field(dtype=ti.i32, shape=()) # 当前帧号
z_min_field = ti.field(dtype=ti.f32, shape=())
z_max_field = ti.field(dtype=ti.f32, shape=())
z_min_field[None] = z_min
z_max_field[None] = z_max
# # ====== 拷贝数据一次 ======
c_start.from_numpy(starts)
c_end.from_numpy(ends)
c_rgba.from_numpy(colors)
f_offset.from_numpy(frame_offset)
f_count.from_numpy(frame_count)
@ti.func
def sd_cylinder(p, a, b, r):
pa = p - a
ba = b - a
h = ba.norm()
eps = 1e-8
res = 0.0
if h < eps:
res = pa.norm() - r
else:
ba_n = ba / h
proj = pa.dot(ba_n)
proj_clamped = min(max(proj, 0.0), h)
res = (pa - proj_clamped * ba_n).norm() - r
return res
@ti.func
def scene_sdf(p):
best_d = 1e6
best_col = ti.Vector([0.0, 0.0, 0.0, 0.0])
fid = frame_id[None] # 从 field 里读出来,变成一个普通 int
off = f_offset[fid]
cnt = f_count[fid]
for i in range(cnt): # 只遍历实际数量
a = c_start[off + i]
b = c_end[off + i]
r = radius
col = c_rgba[off + i]
d = sd_cylinder(p, a, b, r)
if d < best_d:
best_d = d
best_col = col
return best_d, best_col
@ti.func
def get_normal(p):
e = 1e-3
dx = scene_sdf(p + ti.Vector([e, 0.0, 0.0]))[0] - scene_sdf(p - ti.Vector([e, 0.0, 0.0]))[0]
dy = scene_sdf(p + ti.Vector([0.0, e, 0.0]))[0] - scene_sdf(p - ti.Vector([0.0, e, 0.0]))[0]
dz = scene_sdf(p + ti.Vector([0.0, 0.0, e]))[0] - scene_sdf(p - ti.Vector([0.0, 0.0, e]))[0]
n = ti.Vector([dx, dy, dz])
return n.normalized()
@ti.func
def pixel_to_ray(xi, yi):
u = (xi - cx) / fx
v = (yi - cy) / fy
dir_cam = ti.Vector([u, v, 1.0]).normalized()
Rcw = ti.Matrix.identity(ti.f32, 3)
rd_world = Rcw @ dir_cam
ro_world = C
return ro_world, rd_world
@ti.kernel
def render():
depth_near, depth_far = ti.max(z_min_field[None], 0.1), ti.min(z_max_field[None] + 6000, 20000) # 能渲染出来的点,最大12000
for y, x in img:
ro, rd = pixel_to_ray(x, y)
t = znear
col_out = ti.Vector([0.0, 0.0, 0.0, 0.0])
for _ in range(300):
p = ro + rd * t
d, col = scene_sdf(p)
if d < 1e-3:
# n = get_normal(p)
# diff = max(n.dot(-light_dir), 0.0)
# lit = 0.3 + 0.7 * diff
# col_out = ti.Vector([col.x * lit, col.y * lit, col.z * lit, col.w])
# break
n = get_normal(p)
diff = max(n.dot(-light_dir), 0.0)
# === Blinn-Phong 镜面反射 ===
view_dir = -rd.normalized()
half_dir = (view_dir + -light_dir).normalized()
spec = max(n.dot(half_dir), 0.0) ** 32 # shininess=32,越小越散,越大越锐
depth_factor = 1.0 - (p.z - depth_near) / (depth_far - znear)
depth_factor = ti.max(0.0, ti.min(1.0, depth_factor))
# 原来的 diffuse/ambient 光照
diffuse_term = 0.3 + 0.7 * diff
base = col.xyz * diffuse_term * depth_factor
# 镜面高光(叠加到原有结果上)
highlight = ti.Vector([1.0, 1.0, 1.0]) * (0.5 * spec) * depth_factor
col_out = ti.Vector([base.x + highlight.x,
base.y + highlight.y,
base.z + highlight.z,
col.w])
break
if t > zfar:
break
t += max(d, 1e-4)
img[y, x] = col_out
frames_np_rgba = []
for f in range(len(specs_list)):
# start_time = time.time()
frame_id[None] = f
render()
arr = np.clip(img.to_numpy(), 0, 1)
# end_time = time.time()
# print(f"Frame {f} time: {end_time - start_time} seconds")
arr8 = (arr * 255).astype(np.uint8)
frames_np_rgba.append(arr8)
return frames_np_rgba
def random_cylinder():
"""生成一根随机圆柱 (start, end, color)。"""
# 起点 [-200,200]^2, z 在 [-300,-100]
ax = random.uniform(-200, 200)
ay = random.uniform(-200, 200)
az = random.uniform(300, 400)
start = [ax, ay, az]
# 随机方向和长度
theta = random.uniform(0, 2*math.pi)
phi = random.uniform(-math.pi/4, math.pi/4) # 倾斜角
L = 100
dx = math.cos(phi) * math.cos(theta)
dy = math.cos(phi) * math.sin(theta)
dz = math.sin(phi)
end = [ax + dx * L, ay + dy * L, az + dz * L]
# 随机颜色 (RGB + alpha=1)
color = [random.random(), random.random(), random.random(), 1.0]
return (start, end, color)
def generate_specs_list(num_frames=120, min_cyl=10, max_cyl=120):
"""生成 specs_list,每帧有若干随机圆柱."""
specs_list = []
for _ in range(num_frames):
n_cyl = random.randint(min_cyl, max_cyl)
specs = [random_cylinder() for _ in range(n_cyl)]
specs_x_shift = [([spec[0][0] + 50, spec[0][1], spec[0][2]], [spec[1][0] + 50, spec[1][1], spec[1][2]], spec[2]) for spec in specs]
specs_list.append(specs)
specs_list.append(specs_x_shift)
return specs_list
+6
View File
@@ -0,0 +1,6 @@
taichi
pyrender
trimesh
opencv-python
matplotlib
pillow
+191
View File
@@ -0,0 +1,191 @@
import numpy as np
import cv2
# source
# https://github.com/Wan-Video/Wan2.2/blob/e9783574ef77be11fcab9aa5607905402538c08d/wan/modules/animate/preprocess/pose2d_utils.py#L1034
def bbox_from_detector(bbox, input_resolution=(224, 224), rescale=1.25):
"""
Get center and scale of bounding box from bounding box.
The expected format is [min_x, min_y, max_x, max_y].
"""
CROP_IMG_HEIGHT, CROP_IMG_WIDTH = input_resolution
CROP_ASPECT_RATIO = CROP_IMG_HEIGHT / float(CROP_IMG_WIDTH)
# center
center_x = (bbox[0] + bbox[2]) / 2.0
center_y = (bbox[1] + bbox[3]) / 2.0
center = np.array([center_x, center_y])
# scale
bbox_w = bbox[2] - bbox[0]
bbox_h = bbox[3] - bbox[1]
bbox_size = max(bbox_w * CROP_ASPECT_RATIO, bbox_h)
scale = np.array([bbox_size / CROP_ASPECT_RATIO, bbox_size]) / 200.0
# scale = bbox_size / 200.0
# adjust bounding box tightness
scale *= rescale
return center, scale
def get_transform(center, scale, res, rot=0):
"""Generate transformation matrix."""
# res: (height, width), (rows, cols)
crop_aspect_ratio = res[0] / float(res[1])
h = 200 * scale
w = h / crop_aspect_ratio
t = np.zeros((3, 3))
t[0, 0] = float(res[1]) / w
t[1, 1] = float(res[0]) / h
t[0, 2] = res[1] * (-float(center[0]) / w + .5)
t[1, 2] = res[0] * (-float(center[1]) / h + .5)
t[2, 2] = 1
if not rot == 0:
rot = -rot # To match direction of rotation from cropping
rot_mat = np.zeros((3, 3))
rot_rad = rot * np.pi / 180
sn, cs = np.sin(rot_rad), np.cos(rot_rad)
rot_mat[0, :2] = [cs, -sn]
rot_mat[1, :2] = [sn, cs]
rot_mat[2, 2] = 1
# Need to rotate around center
t_mat = np.eye(3)
t_mat[0, 2] = -res[1] / 2
t_mat[1, 2] = -res[0] / 2
t_inv = t_mat.copy()
t_inv[:2, 2] *= -1
t = np.dot(t_inv, np.dot(rot_mat, np.dot(t_mat, t)))
return t
def transform(pt, center, scale, res, invert=0, rot=0):
"""Transform pixel location to different reference."""
t = get_transform(center, scale, res, rot=rot)
if invert:
t = np.linalg.inv(t)
new_pt = np.array([pt[0] - 1, pt[1] - 1, 1.]).T
new_pt = np.dot(t, new_pt)
return np.array([round(new_pt[0]), round(new_pt[1])], dtype=int) + 1
def crop(img, center, scale, res):
"""
Crop image according to the supplied bounding box.
res: [rows, cols]
"""
# Upper left point
ul = np.array(transform([1, 1], center, max(scale), res, invert=1)) - 1
# Bottom right point
br = np.array(transform([res[1] + 1, res[0] + 1], center, max(scale), res, invert=1)) - 1
new_shape = [br[1] - ul[1], br[0] - ul[0]]
if len(img.shape) > 2:
new_shape += [img.shape[2]]
new_img = np.zeros(new_shape, dtype=np.float32)
# Range to fill new array
new_x = max(0, -ul[0]), min(br[0], len(img[0])) - ul[0]
new_y = max(0, -ul[1]), min(br[1], len(img)) - ul[1]
# Range to sample from original image
old_x = max(0, ul[0]), min(len(img[0]), br[0])
old_y = max(0, ul[1]), min(len(img), br[1])
try:
new_img[new_y[0]:new_y[1], new_x[0]:new_x[1]] = img[old_y[0]:old_y[1], old_x[0]:old_x[1]]
except Exception as e:
print(e)
new_img = cv2.resize(new_img, (res[1], res[0])) # (cols, rows)
return new_img, new_shape, (old_x, old_y), (new_x, new_y) # , ul, br
def split_kp2ds_for_aa(kp2ds, ret_face=False):
kp2ds_body = (kp2ds[[0, 6, 6, 8, 10, 5, 7, 9, 12, 14, 16, 11, 13, 15, 2, 1, 4, 3, 17, 20]] + kp2ds[[0, 5, 6, 8, 10, 5, 7, 9, 12, 14, 16, 11, 13, 15, 2, 1, 4, 3, 18, 21]]) / 2
kp2ds_lhand = kp2ds[91:112]
kp2ds_rhand = kp2ds[112:133]
kp2ds_face = kp2ds[22:91]
if ret_face:
return kp2ds_body.copy(), kp2ds_lhand.copy(), kp2ds_rhand.copy(), kp2ds_face.copy()
return kp2ds_body.copy(), kp2ds_lhand.copy(), kp2ds_rhand.copy()
def load_pose_metas_from_kp2ds_seq(kp2ds_seq, width, height):
metas = []
last_kp2ds_body = None
for kps in kp2ds_seq:
kps = kps.copy()
kps[:, 0] /= width
kps[:, 1] /= height
kp2ds_body, kp2ds_lhand, kp2ds_rhand, kp2ds_face = split_kp2ds_for_aa(kps, ret_face=True)
# Exclude cases where all values are less than 0
if last_kp2ds_body is not None and kp2ds_body[:, :2].min(axis=1).max() < 0:
kp2ds_body = last_kp2ds_body
last_kp2ds_body = kp2ds_body
meta = {
"width": width,
"height": height,
"keypoints_body": kp2ds_body,
"keypoints_left_hand": kp2ds_lhand,
"keypoints_right_hand": kp2ds_rhand,
"keypoints_face": kp2ds_face,
}
metas.append(meta)
return metas
def aaposemeta_to_dwpose_scail(meta):
"""
Convert AA pose metadata to DWpose format matching DWposeDetector output.
DWpose format:
- bodies: dict with 'candidate' (n, 24, 2) and 'subset' (n, 24) where subset contains indices
- hands: array (2*n, 21, 2) - stacked right/left hands
- faces: array (n, 68, 2)
"""
# Body keypoints (excluding last 2)
candidate_body = meta['keypoints_body'][:-2][:, :2] # (24, 2)
score_body = meta['keypoints_body'][:-2][:, 2] # (24,)
# Create subset: contains joint index if visible, -1 if not
subset_body = np.arange(len(candidate_body), dtype=float)
subset_body[score_body <= 0.3] = -1 # Match DWpose threshold
# Bodies dict with single person (expand to match multi-person format)
bodies = {
"candidate": np.expand_dims(candidate_body, axis=0), # (1, 24, 2)
"subset": np.expand_dims(subset_body, axis=0) # (1, 24)
}
# Hands: stack right then left (2, 21, 2)
hands_coords = np.stack([
meta['keypoints_right_hand'][:, :2],
meta['keypoints_left_hand'][:, :2]
], axis=0)
hands_score = np.stack([
meta['keypoints_right_hand'][:, 2],
meta['keypoints_left_hand'][:, 2]
], axis=0)
# Faces: (1, 68, 2) - skip first face keypoint like DWpose does (24:92 = 68 points)
faces_coords = np.expand_dims(meta['keypoints_face'][1:][:, :2], axis=0)
faces_score = np.expand_dims(meta['keypoints_face'][1:][:, 2], axis=0)
# Match DWpose output structure
dwpose_format = {
"bodies": bodies,
"hands": hands_coords,
"faces": faces_coords
}
# Optional: include scores separately like DWpose does
score_dict = {
"body_score": np.expand_dims(score_body, axis=0),
"hand_score": hands_score,
"face_score": faces_score
}
# Merge score dict into dwpose_format
dwpose_format.update(score_dict)
return dwpose_format