fixed the relative depth conversion, don't feed the math gremlins after midnight

This commit is contained in:
spacepxl
2024-10-04 23:53:01 -04:00
parent 147795818b
commit 8b03754d91
4 changed files with 24 additions and 10 deletions
+1 -1
View File
@@ -6,7 +6,7 @@ Based on https://github.com/apple/ml-depth-pro
The raw output of the depth model is metric depth (aka, distance from camera in meters) which may have values up in the hundreds or thousands for far away objects. This is great for projection to 3d, and you can use the focal length estimate to make a camera (focal_mm = focal_px * sensor_mm / sensor_px)
In order to convert metric depth to relative depth, like what's needed for controlnet, the depth has to be remapped into the 0 to 1 range, which seems simple enough but is actually quite subjective. Try adjusting the `std_dev` (rejects extreme outlier values) and `gamma` (an exponent to bias the image brighter or darker) depending on the image or video.
In order to convert metric depth to relative depth, like what's needed for controlnet, the depth has to be remapped into the 0 to 1 range, which is handled by a separate node. The defaults should be good for most uses, but you can invert it and/or use `gamma` to bias it brighter or darker.
If you get errors about "vit_large_patch14_dinov2" make sure timm is up to date (tested with 0.9.16 and 1.0.9)
Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.2 MiB

After

Width:  |  Height:  |  Size: 1.2 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 1.4 MiB

+23 -9
View File
@@ -101,9 +101,8 @@ class MetricDepthToRelative:
return {
"required": {
"depth": ("IMAGE",),
"per_image": ("BOOLEAN", {"default": False,}),
"per_image": ("BOOLEAN", {"default": True,}),
"invert": ("BOOLEAN", {"default": True,}),
"std_dev": ("FLOAT", {"default": 5.0, "min": 0.1, "max": 100.0, "step": 0.1, "round": 0.1}),
"gamma": ("FLOAT", {"default": 1.0, "min": 0.01, "max": 100, "step": 0.01}),
},
}
@@ -113,22 +112,18 @@ class MetricDepthToRelative:
FUNCTION = "convert_depth"
CATEGORY = "Depth-Pro"
def convert_depth(self, depth, per_image, invert, std_dev, gamma):
relative_depth = depth.detach().clone()
def convert_depth(self, depth, per_image, invert, gamma):
relative_depth = 1 / (1 + depth.detach().clone())
if per_image:
for i in range(relative_depth.size(0)):
std, mean = torch.std_mean(relative_depth[i], dim=None)
relative_depth[i] = torch.clamp(relative_depth[i], min = 0, max = std * std_dev + mean)
relative_depth[i] = relative_depth[i] - relative_depth[i].min()
relative_depth[i] = relative_depth[i] / relative_depth[i].max()
else:
std, mean = torch.std_mean(relative_depth, dim=None)
relative_depth = torch.clamp(relative_depth, min = 0, max = std * std_dev + mean)
relative_depth = relative_depth - relative_depth.min()
relative_depth = relative_depth / relative_depth.max()
if invert:
if not invert:
relative_depth = 1 - relative_depth
if gamma != 1:
@@ -137,14 +132,33 @@ class MetricDepthToRelative:
return (relative_depth,)
class MetricDepthToInverse:
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"depth": ("IMAGE",),
},
}
RETURN_TYPES = ("IMAGE",)
RETURN_NAMES = ("depth",)
FUNCTION = "convert_depth"
CATEGORY = "Depth-Pro"
def convert_depth(self, depth):
return (1 / (1 + depth.detach().clone()), )
NODE_CLASS_MAPPINGS = {
"LoadDepthPro": LoadDepthPro,
"DepthPro": DepthPro,
"MetricDepthToRelative": MetricDepthToRelative,
"MetricDepthToInverse": MetricDepthToInverse,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"LoadDepthPro": "(Down)Load Depth Pro model",
"DepthPro": "Depth Pro",
"MetricDepthToRelative": "Metric Depth to Relative",
"MetricDepthToInverse": "Metric Depth to Inverse",
}