further cleaning the image lib split

support for non-alpha comfy load image image (flat alpha)
This commit is contained in:
Alexander G. Morano
2024-09-18 14:47:51 -07:00
parent b18d1e655e
commit ed67ec57c5
20 changed files with 1592 additions and 1510 deletions
+24 -18
View File
@@ -20,26 +20,32 @@ from Jovimetrix import JOV_TYPE_IMAGE, JOVBaseNode, JOVImageNode, Lexicon, \
from Jovimetrix.sup.util import EnumConvertType, parse_dynamic, parse_param, \
zip_longest_fill
from Jovimetrix.sup.image import MIN_IMAGE_SIZE, EnumImageType, EnumColorTheory, \
EnumProjection, EnumScaleMode, EnumEdge, EnumMirrorMode, EnumOrientation, \
EnumPixelSwizzle, EnumBlendType, EnumCBDeficiency, EnumCBSimulator, \
EnumColorMap, EnumAdjustOP, EnumThreshold, EnumInterpolation, \
EnumThresholdAdapt, cv2tensor_full, image_blend, image_crop, image_crop_center, image_flatten, \
from Jovimetrix.sup.image import MIN_IMAGE_SIZE, EnumImageType, EnumScaleMode, \
EnumInterpolation, cv2tensor_full, image_blend, image_crop, image_crop_center, \
image_grayscale, image_mask, image_mask_add, image_matte, image_minmax, \
image_scalefit, tensor2cv, cv2tensor, pixel_eval, image_convert, channel_merge, \
channel_solid, channel_swap, image_crop_polygonal
tensor2cv, cv2tensor, pixel_eval, image_convert, image_crop_polygonal
from Jovimetrix.sup.image.color import color_match_lut, color_match_reinhard, \
color_theory, color_blind
from Jovimetrix.sup.image.color import EnumCBDeficiency, EnumCBSimulator, \
EnumColorMap, EnumColorTheory, color_match_lut, color_match_reinhard, \
color_theory, color_blind, image_gradient_map
from Jovimetrix.sup.image.adjust import image_contrast, image_edge_wrap, \
image_equalize, image_filter, image_gamma, image_hsv, image_invert, \
image_transform
from Jovimetrix.sup.image.adjust import EnumEdge, EnumMirrorMode, image_contrast, \
image_edge_wrap, image_equalize, image_filter, image_gamma, image_hsv, \
image_invert, image_mirror, image_pixelate, image_posterize, image_quantize, \
image_scalefit, image_sharpen, image_transform
from Jovimetrix.sup.image.misc import image_gradient_map, image_stack, \
image_mirror, image_threshold, image_quantize, image_levels, \
morph_edge_detect, remap_sphere, image_sharpen, morph_emboss, remap_fisheye, \
remap_perspective, remap_polar, image_split, image_pixelate, image_posterize
from Jovimetrix.sup.image.misc import EnumProjection, EnumThreshold, \
EnumThresholdAdapt, image_threshold, morph_edge_detect, morph_emboss, \
image_split
from Jovimetrix.sup.image.channel import EnumPixelSwizzle, channel_merge, \
channel_solid
from Jovimetrix.sup.image.compose import EnumAdjustOP, EnumBlendType, \
EnumOrientation, image_flatten, image_levels, image_stack
from Jovimetrix.sup.image.mapping import remap_fisheye, remap_perspective, \
remap_polar, remap_sphere
# =============================================================================
@@ -433,7 +439,8 @@ Generate a color harmony based on the selected scheme. Supported schemes include
"optional": {
Lexicon.PIXEL: (JOV_TYPE_IMAGE, {}),
Lexicon.SCHEME: (EnumColorTheory._member_names_, {"default": EnumColorTheory.COMPLIMENTARY.name}),
Lexicon.VALUE: ("INT", {"default": 45, "mij": -90, "maj": 90, "tooltips": "Custom angle of separation to use when calculating colors"}),
Lexicon.VALUE: ("INT", {"default": 45, "mij": -90, "maj": 90,
"tooltips": "Custom angle of separation to use when calculating colors"}),
Lexicon.INVERT: ("BOOLEAN", {"default": False})
}
})
@@ -865,7 +872,6 @@ Swap pixel values between two input images based on specified channel swizzle op
if (who := EnumPixelSwizzle[who]) != EnumPixelSwizzle.CONSTANT:
side = who.value % 10
idx = who.value // 10
print(chan, side, idx, i, who)
out[:,:,i] = (pB if side == 1 else pA)[:,:,chan]
images.append(cv2tensor_full(out))
+10 -8
View File
@@ -19,14 +19,17 @@ from Jovimetrix import JOV_TYPE_IMAGE, JOVBaseNode, JOVImageNode, Lexicon, \
from Jovimetrix.sup.util import EnumConvertType, parse_param, zip_longest_fill
from Jovimetrix.sup.image import MIN_IMAGE_SIZE, EnumScaleMode, EnumInterpolation, \
EnumEdge, EnumImageType, EnumShapes, channel_solid, cv2tensor, cv2tensor_full, \
image_mask_add, image_matte, image_scalefit, tensor2cv, pil2cv
EnumImageType, cv2tensor, cv2tensor_full, image_mask_add, image_matte, \
tensor2cv, pil2cv
from Jovimetrix.sup.image.channel import channel_solid
from Jovimetrix.sup.image.compose import image_mask_binary
from Jovimetrix.sup.image.adjust import image_invert, image_rotate, image_transform, image_translate
from Jovimetrix.sup.image.adjust import EnumEdge, image_invert, image_rotate, \
image_scalefit, image_transform, image_translate
from Jovimetrix.sup.image.misc import image_stereogram, shape_ellipse, \
from Jovimetrix.sup.image.misc import EnumShapes, image_stereogram, shape_ellipse, \
shape_polygon, shape_quad
from Jovimetrix.sup.text import EnumAlignment, EnumJustify, font_names, \
@@ -34,7 +37,6 @@ from Jovimetrix.sup.text import EnumAlignment, EnumJustify, font_names, \
from Jovimetrix.sup.audio import graph_sausage
# =============================================================================
JOV_CATEGORY = "CREATE"
@@ -58,17 +60,17 @@ Generate a constant image or mask of a specified size and color. It can be used
Lexicon.PIXEL: (JOV_TYPE_IMAGE, {"tooltips":"Optional Image to Matte with Selected Color"}),
Lexicon.RGBA_A: ("VEC4INT", {"default": (0, 0, 0, 255),
"rgb": True, "tooltips": "Constant Color to Output"}),
Lexicon.MODE: (EnumScaleMode._member_names_, {"default": EnumScaleMode.MATTE.name}),
Lexicon.WH: ("VEC2INT", {"default": (512, 512),
"label": [Lexicon.W, Lexicon.H],
"tooltips": "Desired Width and Height of the Color Output"}),
Lexicon.MODE: (EnumScaleMode._member_names_, {"default": EnumScaleMode.MATTE.name}),
Lexicon.SAMPLE: (EnumInterpolation._member_names_, {"default": EnumInterpolation.LANCZOS4.name}),
}
})
return Lexicon._parse(d, cls)
def run(self, **kw) -> Tuple[torch.Tensor, torch.Tensor]:
pA = parse_param(kw, Lexicon.PIXEL, EnumConvertType.IMAGE, None)
def run(self, **kw) -> Tuple[torch.Tensor, ...]:
pA = parse_param(kw, Lexicon.PIXEL, EnumConvertType.IMAGE, [None])
matte = parse_param(kw, Lexicon.RGBA_A, EnumConvertType.VEC4INT, [(0, 0, 0, 255)], 0, 255)
wihi = parse_param(kw, Lexicon.WH, EnumConvertType.VEC2INT, [(512, 512)], MIN_IMAGE_SIZE)
mode = parse_param(kw, Lexicon.MODE, EnumConvertType.STRING, EnumScaleMode.MATTE.name)
+4 -1
View File
@@ -10,6 +10,7 @@ from typing import Any, Tuple
import torch
from loguru import logger
try:
from server import PromptServer
from aiohttp import web
@@ -23,8 +24,10 @@ from Jovimetrix import JOV_TYPE_IMAGE, Lexicon, JOVImageNode, \
from Jovimetrix.sup.util import EnumConvertType, parse_param, \
parse_value
from Jovimetrix.sup.image.adjust import image_scalefit
from Jovimetrix.sup.image import MIN_IMAGE_SIZE, EnumInterpolation, \
EnumScaleMode, cv2tensor_full, image_convert, image_scalefit, tensor2cv
EnumScaleMode, cv2tensor_full, image_convert, tensor2cv
import Jovimetrix.sup.shader as glsl_enums
+1 -1
View File
@@ -7,7 +7,7 @@ Device -- MIDI
type 2 (asynchronous): each track is independent of the others
"""
from typing import Any, Tuple
from typing import Tuple
from math import isclose
from queue import Queue
+6 -3
View File
@@ -23,12 +23,15 @@ from Jovimetrix.sup.stream import camera_list, monitor_list, window_list, \
monitor_capture, window_capture, StreamingServer, StreamManager, \
MediaStreamDevice, JOV_SPOUT
from Jovimetrix.sup.image.adjust import image_scalefit
from Jovimetrix.sup.image.channel import channel_solid
if JOV_SPOUT:
from Jovimetrix.sup.stream import SpoutSender, MediaStreamSpout
from Jovimetrix.sup.image import channel_solid, \
cv2tensor_full, image_convert, pixel_eval, tensor2cv, image_scalefit, \
EnumInterpolation, EnumScaleMode, EnumImageType, MIN_IMAGE_SIZE
from Jovimetrix.sup.image import cv2tensor_full, image_convert, pixel_eval, \
tensor2cv, EnumInterpolation, EnumScaleMode, EnumImageType, MIN_IMAGE_SIZE
# =============================================================================
+6 -4
View File
@@ -21,16 +21,18 @@ from loguru import logger
from comfy.utils import ProgressBar
from nodes import interrupt_processing
from Jovimetrix import JOV_TYPE_ANY, ROOT, \
Lexicon, JOVBaseNode, deep_merge, comfy_message, parse_reset
from Jovimetrix import JOV_TYPE_ANY, ROOT, Lexicon, JOVBaseNode, deep_merge, \
comfy_message, parse_reset
from Jovimetrix.sup.util import EnumConvertType, parse_dynamic, parse_param
from Jovimetrix.sup.image import MIN_IMAGE_SIZE, IMAGE_FORMATS, EnumInterpolation, \
EnumScaleMode, cv2tensor, cv2tensor_full, image_convert, \
image_matte, image_scalefit, tensor2cv, image_load
image_matte, tensor2cv, image_load
from Jovimetrix.sup.image.misc import image_by_size
from Jovimetrix.sup.image.adjust import image_scalefit
from Jovimetrix.sup.image.compose import image_by_size
# =============================================================================
+28
View File
@@ -227,6 +227,34 @@ Exports and Displays immediate information about images.
cc = 1
return count, width, height, cc, (width, height), (width, height, cc)
class Passthru(JOVBaseNode):
NAME = "PASSTHRU (JOV) 🚌"
CATEGORY = f"JOVIMETRIX 🔺🟩🔵/{JOV_CATEGORY}"
RETURN_TYPES = ()
RETURN_NAMES = ()
SORT = 860
DESCRIPTION = """
Passes the data into python so it can be probed.
"""
OUTPUT_NODE = True
@classmethod
def INPUT_TYPES(cls) -> dict:
d = super().INPUT_TYPES()
d = deep_merge(d, {
"optional": {
Lexicon.UNKNOWN: (JOV_TYPE_ANY, {"default": None, "tooltips":"Pass through data."}),
}
})
return Lexicon._parse(d, cls)
def run(self, **kw) -> Tuple[Any, ...]:
inout = parse_param(kw, Lexicon.UNKNOWN, EnumConvertType.ANY, [None])
for x in inout:
logger.info(f"{type(x)}")
# logger.info(dir(x))
return ()
class RouteNode(JOVBaseNode):
NAME = "ROUTE (JOV) 🚌"
CATEGORY = f"JOVIMETRIX 🔺🟩🔵/{JOV_CATEGORY}"
+3 -4
View File
@@ -19,11 +19,10 @@ from loguru import logger
from comfy.utils import ProgressBar
from folder_paths import get_output_directory
from Jovimetrix import JOV_TYPE_ANY, JOV_TYPE_IMAGE, \
Lexicon, JOVBaseNode, deep_merge
from Jovimetrix import JOV_TYPE_IMAGE, Lexicon, JOVBaseNode, deep_merge
from Jovimetrix.sup.util import EnumConvertType, \
path_next, parse_param, zip_longest_fill
from Jovimetrix.sup.util import EnumConvertType, path_next, parse_param, \
zip_longest_fill
from Jovimetrix.sup.image import tensor2cv, tensor2pil
+1
View File
@@ -37,6 +37,7 @@
"MIDI READER (JOV) \ud83c\udfb9": "Captures MIDI messages from an external MIDI device or controller",
"OP BINARY (JOV) \ud83c\udf1f": "Execute binary operations like addition, subtraction, multiplication, division, and bitwise operations on input values, supporting various data types and vector sizes",
"OP UNARY (JOV) \ud83c\udfb2": "Perform single function operations like absolute value, mean, median, mode, magnitude, normalization, maximum, or minimum on input values",
"PASSTHRU (JOV) \ud83d\ude8c": "Passes the data into python so it can be probed",
"PIXEL MERGE (JOV) \ud83e\udec2": "Combines individual color channels (red, green, blue) along with an optional mask channel to create a composite image",
"PIXEL SPLIT (JOV) \ud83d\udc94": "Takes an input image and splits it into its individual color channels (red, green, blue), along with a mask channel",
"PIXEL SWAP (JOV) \ud83d\udd03": "Swap pixel values between two input images based on specified channel swizzle operations",
+3 -1
View File
@@ -11,7 +11,9 @@ from PIL import Image, ImageDraw
from loguru import logger
from Jovimetrix.sup.image import TYPE_PIXEL, EnumImageType, EnumScaleMode, \
pixel_eval, image_scalefit, pil2cv
pixel_eval, pil2cv
from Jovimetrix.sup.image.adjust import image_scalefit
# =============================================================================
+105 -609
View File
@@ -13,18 +13,15 @@
"""
import io
from io import BytesIO
import math
import base64
import urllib
import requests
from enum import Enum
from io import BytesIO
from typing import List, Optional, Tuple
import cv2
import torch
import numpy as np
from daltonlens import simulate
from PIL import Image, ImageOps
from blendmodes.blend import BlendType, blendLayers
@@ -63,112 +60,6 @@ TYPE_VECTOR = TYPE_IMAGE | TYPE_PIXEL
# === ENUMERATION ===
# =============================================================================
class EnumAdjustOP(Enum):
BLUR = 0
STACK_BLUR = 1
GAUSSIAN_BLUR = 2
MEDIAN_BLUR = 3
SHARPEN = 10
EMBOSS = 20
INVERT = 25
# MEAN = 30 -- in UNARY
# ADAPTIVE_HISTOGRAM = 35
HSV = 30
LEVELS = 35
EQUALIZE = 40
PIXELATE = 50
QUANTIZE = 55
POSTERIZE = 60
FIND_EDGES = 80
OUTLINE = 70
DILATE = 71
ERODE = 72
OPEN = 73
CLOSE = 74
class EnumBlendType(Enum):
"""Rename the blend type names."""
NORMAL = BlendType.NORMAL
ADDITIVE = BlendType.ADDITIVE
NEGATION = BlendType.NEGATION
DIFFERENCE = BlendType.DIFFERENCE
MULTIPLY = BlendType.MULTIPLY
DIVIDE = BlendType.DIVIDE
LIGHTEN = BlendType.LIGHTEN
DARKEN = BlendType.DARKEN
SCREEN = BlendType.SCREEN
BURN = BlendType.COLOURBURN
DODGE = BlendType.COLOURDODGE
OVERLAY = BlendType.OVERLAY
HUE = BlendType.HUE
SATURATION = BlendType.SATURATION
LUMINOSITY = BlendType.LUMINOSITY
COLOR = BlendType.COLOUR
SOFT = BlendType.SOFTLIGHT
HARD = BlendType.HARDLIGHT
PIN = BlendType.PINLIGHT
VIVID = BlendType.VIVIDLIGHT
EXCLUSION = BlendType.EXCLUSION
REFLECT = BlendType.REFLECT
GLOW = BlendType.GLOW
XOR = BlendType.XOR
EXTRACT = BlendType.GRAINEXTRACT
MERGE = BlendType.GRAINMERGE
DESTIN = BlendType.DESTIN
DESTOUT = BlendType.DESTOUT
SRCATOP = BlendType.SRCATOP
DESTATOP = BlendType.DESTATOP
class EnumImageBySize(Enum):
LARGEST = 10
SMALLEST = 20
WIDTH_MIN = 30
WIDTH_MAX = 40
HEIGHT_MIN = 50
HEIGHT_MAX = 60
class EnumColorMap(Enum):
AUTUMN = cv2.COLORMAP_AUTUMN
BONE = cv2.COLORMAP_BONE
JET = cv2.COLORMAP_JET
WINTER = cv2.COLORMAP_WINTER
RAINBOW = cv2.COLORMAP_RAINBOW
OCEAN = cv2.COLORMAP_OCEAN
SUMMER = cv2.COLORMAP_SUMMER
SPRING = cv2.COLORMAP_SPRING
COOL = cv2.COLORMAP_COOL
HSV = cv2.COLORMAP_HSV
PINK = cv2.COLORMAP_PINK
HOT = cv2.COLORMAP_HOT
PARULA = cv2.COLORMAP_PARULA
MAGMA = cv2.COLORMAP_MAGMA
INFERNO = cv2.COLORMAP_INFERNO
PLASMA = cv2.COLORMAP_PLASMA
VIRIDIS = cv2.COLORMAP_VIRIDIS
CIVIDIS = cv2.COLORMAP_CIVIDIS
TWILIGHT = cv2.COLORMAP_TWILIGHT
TWILIGHT_SHIFTED = cv2.COLORMAP_TWILIGHT_SHIFTED
TURBO = cv2.COLORMAP_TURBO
DEEPGREEN = cv2.COLORMAP_DEEPGREEN
class EnumColorTheory(Enum):
COMPLIMENTARY = 0
MONOCHROMATIC = 1
SPLIT_COMPLIMENTARY = 2
ANALOGOUS = 3
TRIADIC = 4
# TETRADIC = 5
SQUARE = 6
COMPOUND = 8
# DOUBLE_COMPLIMENTARY = 9
CUSTOM_TETRAD = 9
class EnumEdge(Enum):
CLIP = 1
WRAP = 2
WRAPX = 3
WRAPY = 4
class EnumGrayscaleCrunch(Enum):
LOW = 0
HIGH = 1
@@ -197,29 +88,6 @@ class EnumIntFloat(Enum):
FLOAT = 0
INT = 1
class EnumMirrorMode(Enum):
NONE = -1
X = 0
FLIP_X = 10
Y = 20
FLIP_Y = 30
XY = 40
X_FLIP_Y = 50
FLIP_XY = 60
FLIP_X_FLIP_Y = 70
class EnumOrientation(Enum):
HORIZONTAL = 0
VERTICAL = 1
GRID = 2
class EnumProjection(Enum):
NORMAL = 0
POLAR = 5
SPHERICAL = 10
FISHEYE = 15
PERSPECTIVE = 20
class EnumScaleMode(Enum):
# NONE = 0
MATTE = 0
@@ -228,174 +96,9 @@ class EnumScaleMode(Enum):
ASPECT = 30
ASPECT_SHORT = 35
class EnumShapes(Enum):
CIRCLE = 0
SQUARE = 1
ELLIPSE = 2
RECTANGLE = 3
POLYGON = 4
class EnumThreshold(Enum):
BINARY = cv2.THRESH_BINARY
TRUNC = cv2.THRESH_TRUNC
TOZERO = cv2.THRESH_TOZERO
class EnumThresholdAdapt(Enum):
ADAPT_NONE = -1
ADAPT_MEAN = cv2.ADAPTIVE_THRESH_MEAN_C
ADAPT_GAUSS = cv2.ADAPTIVE_THRESH_GAUSSIAN_C
class EnumPixelSwizzle(Enum):
RED_A = 20
GREEN_A = 10
BLUE_A = 0
ALPHA_A = 30
RED_B = 21
GREEN_B = 11
BLUE_B = 1
ALPHA_B = 31
CONSTANT = 50
class EnumCBSimulator(Enum):
AUTOSELECT = 0
BRETTEL1997 = 1
COBLISV1 = 2
COBLISV2 = 3
MACHADO2009 = 4
VIENOT1999 = 5
VISCHECK = 6
class EnumCBDeficiency(Enum):
PROTAN = simulate.Deficiency.PROTAN
DEUTAN = simulate.Deficiency.DEUTAN
TRITAN = simulate.Deficiency.TRITAN
# =============================================================================
# === FILE I/O ===
# =============================================================================
def image_load(url: str) -> Tuple[TYPE_IMAGE, ...]:
try:
img = cv2.imread(url, cv2.IMREAD_UNCHANGED)
if img is None:
raise ValueError(f"{url} could not be loaded.")
img = image_normalize(img)
# logger.debug(f"load image {url}: {img.ndim} {img.shape}")
if img.ndim == 3:
if img.shape[2] == 4:
img = cv2.cvtColor(img, cv2.COLOR_RGBA2BGRA)
else:
img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)
elif img.ndim < 3:
img = np.expand_dims(img, -1)
except Exception:
logger.debug(f"load image fallback to PIL {url}")
try:
img = Image.open(url)
img = ImageOps.exif_transpose(img)
img = np.array(img)
if img.dtype != np.uint8:
img = np.clip(np.array(img * 255), 0, 255).astype(dtype=np.uint8)
except Exception as e:
logger.error(str(e))
raise Exception(f"Error loading image: {e}")
if img is None:
raise Exception(f"No file found at {url}")
mask = image_mask(img)
if img.ndim == 3 and img.shape[2] == 4:
img = image_blend(img, img, mask)
img[:,:,3] = mask
return img, mask
def image_load_data(data: str) -> TYPE_IMAGE:
img = ImageOps.exif_transpose(data)
return pil2cv(img)
def image_load_exr(url: str) -> Tuple[TYPE_IMAGE, TYPE_IMAGE]:
"""
exr_file = OpenEXR.InputFile(url)
exr_header = exr_file.header()
r,g,b = exr_file.channels("RGB", pixel_type=Imath.PixelType(Imath.PixelType.FLOAT) )
dw = exr_header[ "dataWindow" ]
w = dw.max.x - dw.min.x + 1
h = dw.max.y - dw.min.y + 1
image = np.ones( (h, w, 4), dtype = np.float32 )
image[:, :, 0] = np.core.multiarray.frombuffer( r, dtype = np.float32 ).reshape(h, w)
image[:, :, 1] = np.core.multiarray.frombuffer( g, dtype = np.float32 ).reshape(h, w)
image[:, :, 2] = np.core.multiarray.frombuffer( b, dtype = np.float32 ).reshape(h, w)
return create_optix_image_2D( w, h, image.flatten() )
"""
pass
def image_load_from_url(url: str) -> TYPE_IMAGE:
"""Creates a CV2 BGR image from a url."""
try:
image = urllib.request.urlopen(url)
image = np.asarray(bytearray(image.read()), dtype=np.uint8)
return cv2.imdecode(image, cv2.IMREAD_COLOR)
except:
try:
image = Image.open(requests.get(url, stream=True).raw)
return pil2cv(image)
except Exception as e:
logger.error(str(e))
def image_save_gif(fpath:str, images: List[Image.Image], fps: int=0,
loop:int=0, optimize:bool=False) -> None:
fps = min(50, max(1, fps))
images[0].save(
fpath,
append_images=images[1:],
duration=3, # int(100.0 / fps),
loop=loop,
optimize=optimize,
save_all=True
)
# =============================================================================
# === CV2 CONVERSION ===
# =============================================================================
MODE_CV2 = {
EnumImageType.BGRA: {
4: cv2.COLOR_RGBA2BGRA,
3: cv2.COLOR_RGB2BGRA,
1: cv2.COLOR_GRAY2BGRA,
},
EnumImageType.RGBA: {
4: lambda x: x,
3: cv2.COLOR_RGB2RGBA,
1: cv2.COLOR_GRAY2RGBA,
},
EnumImageType.BGR: {
4: cv2.COLOR_RGBA2BGR,
3: cv2.COLOR_RGB2BGR,
1: cv2.COLOR_GRAY2BGR,
},
EnumImageType.RGB: {
4: cv2.COLOR_RGBA2RGB,
3: lambda x: x,
1: cv2.COLOR_GRAY2RGB,
},
EnumImageType.GRAYSCALE: {
4: cv2.COLOR_RGBA2GRAY,
3: cv2.COLOR_RGB2GRAY,
1: lambda x: x,
}
}
# =============================================================================
# ==============================================================================
# === CONVERSION ===
# =============================================================================
# ==============================================================================
def bgr2hsv(bgr_color: TYPE_PIXEL) -> TYPE_PIXEL:
return cv2.cvtColor(np.uint8([[bgr_color]]), cv2.COLOR_BGR2HSV)[0, 0]
@@ -447,9 +150,9 @@ def cv2tensor(image: np.ndarray, mask: bool = False) -> torch.Tensor:
return torch.from_numpy(image).unsqueeze(0)
def cv2tensor_full(image: TYPE_IMAGE, matte:TYPE_PIXEL=0) -> Tuple[torch.Tensor, ...]:
rgba = image_convert(image, 4)
rgb = image_matte(rgba, matte)[:,:,:3]
mask = image_mask(rgba)
rgba = image_convert(image, 4, matte=matte)
rgb = image_matte(image, matte)[:,:,:3]
mask = image_mask(image)
rgba = torch.from_numpy(rgba.astype(np.float32) / 255.0).unsqueeze(0)
rgb = torch.from_numpy(rgb.astype(np.float32) / 255.0).unsqueeze(0)
mask = torch.from_numpy(mask.astype(np.float32) / 255.0).unsqueeze(0)
@@ -498,17 +201,7 @@ def tensor2cv(tensor: torch.Tensor) -> TYPE_IMAGE:
tensor = tensor.squeeze()
tensor = tensor.cpu().numpy()
image = np.clip(255.0 * tensor, 0, 255).astype(np.uint8)
"""
if image.shape[2] == 4:
mask = image_mask(image)
# we should flatten against black?
black = np.zeros(image.shape, dtype=np.uint8)
image = image_blend(black, image, mask)
image = image_mask_add(image, mask)
"""
return image
return np.clip(255.0 * tensor, 0, 255).astype(np.uint8)
def tensor2pil(tensor: torch.Tensor) -> Image.Image:
"""Convert a torch Tensor to a PIL Image.
@@ -529,71 +222,9 @@ def mixlabLayer2cv(layer: dict) -> torch.Tensor:
mask = tensor2cv(mask)
return image_mask_add(image, mask)
# =============================================================================
# === COLOR SPACE CONVERSION ===
# =============================================================================
def gamma2linear(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Gamma correction for old PCs/CRT monitors"""
return np.power(image, 2.2)
def linear2gamma(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Inverse gamma correction for old PCs/CRT monitors"""
return np.power(np.clip(image, 0., 1.), 1.0 / 2.2)
def sRGB2Linear(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Convert sRGB to linearRGB, removing the gamma correction.
Works for grayscale, RGB, or RGBA images.
"""
image = image.astype(float) / 255.0
# If the image has an alpha channel, separate it
if image.shape[-1] == 4:
rgb = image[..., :3]
alpha = image[..., 3]
else:
rgb = image
alpha = None
gamma = ((rgb + 0.055) / 1.055) ** 2.4
scale = rgb / 12.92
rgb = np.where(rgb > 0.04045, gamma, scale)
# Recombine the alpha channel if it exists
if alpha is not None:
image = np.concatenate((rgb, alpha[..., np.newaxis]), axis=-1)
else:
image = rgb
return (image * 255).astype(np.uint8)
def linear2sRGB(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Convert linearRGB to sRGB, applying the gamma correction.
Works for grayscale, RGB, or RGBA images.
"""
image = image.astype(float) / 255.0
# If the image has an alpha channel, separate it
if image.shape[-1] == 4:
rgb = image[..., :3]
alpha = image[..., 3]
else:
rgb = image
alpha = None
higher = 1.055 * np.power(rgb, 1.0 / 2.4) - 0.055
lower = rgb * 12.92
rgb = np.where(rgb > 0.0031308, higher, lower)
# Recombine the alpha channel if it exists
if alpha is not None:
image = np.concatenate((rgb, alpha[..., np.newaxis]), axis=-1)
else:
image = rgb
return np.clip(image * 255.0, 0, 255).astype(np.uint8)
# =============================================================================
# ==============================================================================
# === PIXEL ===
# =============================================================================
# ==============================================================================
def pixel_eval(color: TYPE_PIXEL,
target: EnumImageType=EnumImageType.BGR,
@@ -652,152 +283,9 @@ def pixel_eval(color: TYPE_PIXEL,
color = tuple(color[2::-1]) + tuple([color[-1]])
return color
def pixel_hsv_adjust(color:TYPE_PIXEL, hue:int=0, saturation:int=0, value:int=0, mod_color:bool=True, mod_sat:bool=False, mod_value:bool=False) -> TYPE_PIXEL:
"""Adjust an HSV type pixel.
OpenCV uses... H: 0-179, S: 0-255, V: 0-255"""
hsv = [0, 0, 0]
hsv[0] = (color[0] + hue) % 180 if mod_color else np.clip(color[0] + hue, 0, 180)
hsv[1] = (color[1] + saturation) % 255 if mod_sat else np.clip(color[1] + saturation, 0, 255)
hsv[2] = (color[2] + value) % 255 if mod_value else np.clip(color[2] + value, 0, 255)
return hsv
def pixel_convert(color:TYPE_PIXEL, size:int=4, alpha:int=255) -> TYPE_PIXEL:
"""
This function converts X channel pixel into Y channel pixel by adjusting the
size and alpha value if needed.
:param color: The `color` parameter in the `pixel_convert` function represents
the pixel value that you want to convert. It is expected to be a tuple
representing the color channels of the pixel. The number of elements in the
tuple should match the `size` parameter, which specifies the desired number of
color channels
:type color: TYPE_PIXEL
:param size: The `size` parameter in the `pixel_convert` function specifies the
number of channels in the pixel. It determines the expected size of the pixel
tuple that is passed as the `color` argument. The function will modify the
`color` tuple based on the `size` parameter to ensure it matches, defaults to 4
:type size: int (optional)
:param alpha: The `alpha` parameter in the `pixel_convert` function represents
the alpha channel value of the pixel. It is an integer value ranging from 0 to
255, where 0 indicates full transparency and 255 indicates full opacity. The
default value for `alpha` is set to 255 if, defaults to 255
:type alpha: int (optional)
:return: The function `pixel_convert` returns the input `color` if its length is
equal to the specified `size`. If the length of `color` is less than `size`, it
pads the color with zeros to make it of the required length. If `size` is
greater than 2, it adds alpha value to the color if `size` is 4. If `size` is
"""
"""Convert X channel pixel into Y channel pixel."""
if (cc := len(color)) == size:
return color
if size > 2:
color += (0,) * (3 - cc)
if size == 4:
color += (alpha,)
return color
return color[0]
# =============================================================================
# === CHANNEL ===
# =============================================================================
def channel_add(image:TYPE_IMAGE, color:TYPE_PIXEL=255) -> TYPE_IMAGE:
"""
This function adds a new channel with a solid color to an image.
:param image: The `image` parameter is expected to be an image represented as a
NumPy array. The function assumes that the image has a shape attribute that
returns a tuple representing the dimensions of the image (height, width, and
channels if it's a color image)
:type image: TYPE_IMAGE
:param color: The `color` parameter in the `channel_add` function represents the
color value that will be added as a new channel to the input image. The default
value for `color` is 255, which is typically a white color in grayscale images,
defaults to 255
:type color: TYPE_PIXEL (optional)
:return: The function `channel_add` returns a new image with an additional
channel appended to the original image. The new channel has a solid color
specified by the `color` parameter.
"""
h, w = image.shape[:2]
color = pixel_eval(color, EnumImageType.GRAYSCALE)
new = channel_solid(w, h, color, EnumImageType.GRAYSCALE)
return np.concatenate([image, new], axis=-1)
def channel_solid(width:int=MIN_IMAGE_SIZE, height:int=MIN_IMAGE_SIZE, color:TYPE_PIXEL=(0, 0, 0, 255),
chan:EnumImageType=EnumImageType.BGR) -> TYPE_IMAGE:
if chan == EnumImageType.GRAYSCALE:
color = pixel_eval(color, EnumImageType.GRAYSCALE)
what = np.full((height, width, 1), color, dtype=np.uint8)
return what
if not type(color) in [list, set, tuple]:
color = [color]
color += (0,) * (3 - len(color))
if chan in [EnumImageType.BGR, EnumImageType.RGB]:
if chan == EnumImageType.RGB:
color = color[2::-1]
return np.full((height, width, 3), color[:3], dtype=np.uint8)
if len(color) < 4:
color += (255,)
if chan == EnumImageType.RGBA:
color = color[2::-1]
return np.full((height, width, 4), color, dtype=np.uint8)
def channel_merge(channels: List[TYPE_IMAGE]) -> TYPE_IMAGE:
max_height = max(ch.shape[0] for ch in channels if ch is not None)
max_width = max(ch.shape[1] for ch in channels if ch is not None)
num_channels = len(channels)
dtype = channels[0].dtype
output = np.zeros((max_height, max_width, num_channels), dtype=dtype)
for i, channel in enumerate(channels):
if channel is None:
continue
h, w = channel.shape[:2]
if channel.ndim > 2:
channel = channel[..., 0]
pad_top = (max_height - h) // 2
pad_bottom = max_height - h - pad_top
pad_left = (max_width - w) // 2
pad_right = max_width - w - pad_left
padded_channel = np.pad(channel, ((pad_top, pad_bottom), (pad_left, pad_right)),
mode='constant', constant_values=0)
output[..., i] = padded_channel
if num_channels == 1:
output = output[..., 0]
return output
def channel_swap(imageA:TYPE_IMAGE, swap_ot:EnumPixelSwizzle,
imageB:TYPE_IMAGE, swap_in:EnumPixelSwizzle) -> TYPE_IMAGE:
index_out = int(swap_ot.value / 10)
cc_out = imageA.shape[2] if imageA.ndim == 3 else 1
# swap channel is out of range of image size
if index_out > cc_out:
return imageA
index_in = int(swap_in.value / 10)
cc_in = imageB.shape[2] if imageB.ndim == 3 else 1
if index_in > cc_in:
return imageA
imageA[:,:,index_out] = imageB[:,:,index_in]
return imageA
# =============================================================================
# ==============================================================================
# === IMAGE ===
# =============================================================================
"""
These are core functions that most of the support image libraries require.
"""
# ==============================================================================
def image_blend(imageA: TYPE_IMAGE, imageB: TYPE_IMAGE, mask:Optional[TYPE_IMAGE]=None,
blendOp:BlendType=BlendType.NORMAL, alpha:float=1) -> TYPE_IMAGE:
@@ -836,48 +324,48 @@ def image_blend(imageA: TYPE_IMAGE, imageB: TYPE_IMAGE, mask:Optional[TYPE_IMAGE
def image_convert(image: TYPE_IMAGE, channels: int, width: int=None, height: int=None,
matte: Tuple[int, ...]=(0, 0, 0, 0)) -> TYPE_IMAGE:
"""Force image format to a specific number of channels.
Args:
image (TYPE_IMAGE): Input image.
channels (int): Desired number of channels (1, 3, or 4).
width (int): Desired width. `None` means leave unchanged.
height (int): Desired height. `None` means leave unchanged.
matte (tuple): RGBA color to use as background color for transparent areas.
Returns:
TYPE_IMAGE: Image with the specified number of channels.
"""
if image.ndim == 2:
image = np.expand_dims(image, -1)
image = np.expand_dims(image, axis=-1)
cc = image.shape[2]
if cc != channels:
if channels == 1:
image = image[..., :1]
if (cc := image.shape[2]) != channels:
if cc == 1 and channels == 3:
image = np.repeat(image, 3, axis=2)
elif cc == 1 and channels == 4:
rgb = np.repeat(image, 3, axis=2)
alpha = np.full(image.shape[:2] + (1,), matte[3], dtype=image.dtype)
image = np.concatenate([rgb, alpha], axis=2)
elif cc == 3 and channels == 1:
image = np.mean(image, axis=2, keepdims=True).astype(image.dtype)
elif cc == 3 and channels == 4:
alpha = np.full(image.shape[:2] + (1,), matte[3], dtype=image.dtype)
image = np.concatenate([image, alpha], axis=2)
elif cc == 4 and channels == 1:
rgb = image[..., :3]
alpha = image[..., 3:4] / 255.0
image = (np.mean(rgb, axis=2, keepdims=True) * alpha).astype(image.dtype)
elif cc == 4 and channels == 3:
image = image[..., :3]
elif channels == 3:
if cc == 1:
image = np.repeat(image, 3, axis=2)
elif cc == 4:
image = image[..., :3]
elif channels == 4:
if cc == 1:
alpha_channel = np.full(image.shape[:2] + (1,), matte[3], dtype=image.dtype)
image = np.repeat(image, 3, axis=2)
image = np.concatenate((image, alpha_channel), axis=2)
elif cc == 3:
alpha_channel = np.full(image.shape[:2] + (1,), matte[3], dtype=image.dtype)
image = np.concatenate((image, alpha_channel), axis=2)
# If there is expansion, use matte as background and crop/resize if necessary
if width is not None or height is not None:
h, w = image.shape[:2]
width = width or w
height = height or h
image = image_matte(image, matte, width, height)
image = image_crop_center(image, width, height)
# Resize if width or height is specified
h, w = image.shape[:2]
new_width = width if width is not None else w
new_height = height if height is not None else h
if (new_width, new_height) != (w, h):
# Create a new image with the matte color
new_image = np.full((new_height, new_width, channels), matte[:channels], dtype=image.dtype)
paste_x = (new_width - w) // 2
paste_y = (new_height - h) // 2
new_image[paste_y:paste_y+h, paste_x:paste_x+w] = image[:h, :w]
image = new_image
return image
@@ -929,33 +417,6 @@ def image_crop_center(image: TYPE_IMAGE, width:int=None, height:int=None) -> TYP
points = [(x1, y1), (x2, y1), (x2, y2), (x1, y2)]
return image_crop_polygonal(image, points)
def image_flatten(image: List[TYPE_IMAGE], width:int=None, height:int=None,
mode=EnumScaleMode.MATTE,
sample:EnumInterpolation=EnumInterpolation.LANCZOS4) -> TYPE_IMAGE:
if mode == EnumScaleMode.MATTE:
width, height, _, _ = image_minmax(image)[1:]
else:
h, w = image[0].shape[:2]
width = width or w
height = height or h
current = np.zeros((height, width, 4), dtype=np.uint8)
for x in image:
if mode != EnumScaleMode.MATTE:
x = image_scalefit(x, width, height, mode, sample)
x = image_matte(x, (0,0,0,0), width, height)
x = image_scalefit(x, width, height, EnumScaleMode.CROP, sample)
x = image_convert(x, 4)
#@TODO: ADD VARIOUS COMP OPS?
current = cv2.add(current, x)
return current
def image_flatten_mask(image:TYPE_IMAGE, matte:Tuple=(0,0,0,255)) -> Tuple[TYPE_IMAGE, TYPE_IMAGE|None]:
"""Flatten the image with its own alpha channel, if any."""
mask = image_mask(image)
return image_blend(image, image, mask), mask
def image_grayscale(image: TYPE_IMAGE, use_alpha: bool = False) -> TYPE_IMAGE:
"""Convert image to grayscale, optionally using the alpha channel if present.
@@ -979,6 +440,67 @@ def image_grayscale(image: TYPE_IMAGE, use_alpha: bool = False) -> TYPE_IMAGE:
return cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
def image_lerp(imageA: TYPE_IMAGE, imageB:TYPE_IMAGE, mask:TYPE_IMAGE=None,
alpha:float=1.) -> TYPE_IMAGE:
imageA = imageA.astype(np.float32)
imageB = imageB.astype(np.float32)
# establish mask
alpha = np.clip(alpha, 0, 1)
if mask is None:
height, width = imageA.shape[:2]
mask = np.ones((height, width, 1), dtype=np.float32)
else:
# normalize the mask
mask = mask.astype(np.float32)
mask = (mask - mask.min()) / (mask.max() - mask.min()) * alpha
# LERP
imageA = cv2.multiply(1. - mask, imageA)
imageB = cv2.multiply(mask, imageB)
imageA = (cv2.add(imageA, imageB) / 255. - 0.5) * 2.0
imageA = (imageA * 255).astype(np.uint8)
return np.clip(imageA, 0, 255)
def image_load(url: str) -> Tuple[TYPE_IMAGE, ...]:
try:
img = cv2.imread(url, cv2.IMREAD_UNCHANGED)
if img is None:
raise ValueError(f"{url} could not be loaded.")
img = image_normalize(img)
# logger.debug(f"load image {url}: {img.ndim} {img.shape}")
if img.ndim == 3:
if img.shape[2] == 4:
img = cv2.cvtColor(img, cv2.COLOR_RGBA2BGRA)
else:
img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)
elif img.ndim < 3:
img = np.expand_dims(img, -1)
except Exception:
logger.debug(f"load image fallback to PIL {url}")
try:
img = Image.open(url)
img = ImageOps.exif_transpose(img)
img = np.array(img)
if img.dtype != np.uint8:
img = np.clip(np.array(img * 255), 0, 255).astype(dtype=np.uint8)
except Exception as e:
logger.error(str(e))
raise Exception(f"Error loading image: {e}")
if img is None:
raise Exception(f"No file found at {url}")
mask = image_mask(img)
if img.ndim == 3 and img.shape[2] == 4:
img = image_blend(img, img, mask)
img[:,:,3] = mask
return img, mask
def image_mask(image: TYPE_IMAGE, color: TYPE_PIXEL = 255) -> TYPE_IMAGE:
"""Create a mask from the image, preserving transparency.
@@ -1093,6 +615,9 @@ def image_matte(image: TYPE_IMAGE, color: TYPE_iRGBA= (0, 0, 0, 255), width: int
matte[y_offset:y_offset + image_height, x_offset:x_offset + image_width, 3] = image[:, :, 3]
else:
# Handle non-RGBA images (just copy the image onto the matte)
if image.ndim == 2:
image = np.expand_dims(image, axis=-1)
image = np.repeat(image, 3, axis=-1)
matte[y_offset:y_offset + image_height, x_offset:x_offset + image_width, :3] = image[:, :, :3]
return matte
@@ -1120,32 +645,3 @@ def image_normalize(image: TYPE_IMAGE) -> TYPE_IMAGE:
return np.zeros_like(image)
image = (image - img_min) / (img_max - img_min)
return (image * 255).astype(np.uint8)
def image_scalefit(image: TYPE_IMAGE, width: int, height:int,
mode:EnumScaleMode=EnumScaleMode.MATTE,
sample:EnumInterpolation=EnumInterpolation.LANCZOS4,
matte:TYPE_PIXEL=(0,0,0,0)) -> TYPE_IMAGE:
match mode:
case EnumScaleMode.MATTE:
image = image_matte(image, matte, width, height)
case EnumScaleMode.ASPECT:
h, w = image.shape[:2]
ratio = max(width, height) / max(w, h)
image = cv2.resize(image, None, fx=ratio, fy=ratio, interpolation=sample.value)
case EnumScaleMode.ASPECT_SHORT:
h, w = image.shape[:2]
ratio = min(width, height) / min(w, h)
image = cv2.resize(image, None, fx=ratio, fy=ratio, interpolation=sample.value)
case EnumScaleMode.CROP:
image = cv2.resize(image_crop_center(image, width, height), (width, height))
case EnumScaleMode.FIT:
image = cv2.resize(image, (width, height), interpolation=sample.value)
if image.ndim == 2:
image = np.expand_dims(image, -1)
return image
+165 -3
View File
@@ -3,6 +3,7 @@ Jovimetrix - http://www.github.com/amorano/jovimetrix
Support
"""
from enum import Enum
from typing import Tuple
import cv2
@@ -11,10 +12,34 @@ import numpy as np
from loguru import logger
from Jovimetrix.sup.image import TYPE_IMAGE, EnumEdge, EnumInterpolation, \
TYPE_fCOORD2D, bgr2image, cv2tensor, image2bgr, image_crop_center, tensor2cv
from Jovimetrix.sup.image import TYPE_IMAGE, TYPE_PIXEL, EnumInterpolation, \
EnumScaleMode, TYPE_fCOORD2D, bgr2image, cv2tensor, image2bgr, \
image_crop_center, image_matte, tensor2cv
from Jovimetrix.sup.image.misc import image_histogram
# ==============================================================================
# === ENUMERATION ===
# ==============================================================================
class EnumEdge(Enum):
CLIP = 1
WRAP = 2
WRAPX = 3
WRAPY = 4
class EnumMirrorMode(Enum):
NONE = -1
X = 0
FLIP_X = 10
Y = 20
FLIP_Y = 30
XY = 40
X_FLIP_Y = 50
FLIP_XY = 60
FLIP_X_FLIP_Y = 70
# ==============================================================================
# === IMAGE ===
# ==============================================================================
def image_contrast(image: TYPE_IMAGE, value: float) -> TYPE_IMAGE:
image, alpha, cc = image2bgr(image)
@@ -108,6 +133,14 @@ def image_gamma(image: TYPE_IMAGE, value: float) -> TYPE_IMAGE:
# now back to the original "format"
return bgr2image(image, alpha, cc == 1)
def image_histogram(image:TYPE_IMAGE, bins=256) -> TYPE_IMAGE:
bins = max(image.max(), bins) + 1
flatImage = image.flatten()
histogram = np.zeros(bins)
for pixel in flatImage:
histogram[pixel] += 1
return histogram
def image_histogram_normalize(image:TYPE_IMAGE)-> TYPE_IMAGE:
L = image.max()
nonEqualizedHistogram = image_histogram(image, bins=L)
@@ -153,6 +186,89 @@ def image_invert(image: TYPE_IMAGE, value: float) -> TYPE_IMAGE:
inverted_image = 255 - image
return ((1 - value) * image + value * inverted_image).astype(np.uint8)
def image_mirror(image: TYPE_IMAGE, mode:EnumMirrorMode, x:float=0.5,
y:float=0.5) -> TYPE_IMAGE:
cc = image.shape[2] if image.ndim == 3 else 1
height, width = image.shape[:2]
def mirror(img:TYPE_IMAGE, axis:int, reverse:bool=False) -> TYPE_IMAGE:
pivot = x if axis == 1 else y
flip = cv2.flip(img, axis)
pivot = np.clip(pivot, 0, 1)
if reverse:
pivot = 1. - pivot
flip, img = img, flip
scalar = height if axis == 0 else width
slice1 = int(pivot * scalar)
slice1w = scalar - slice1
slice2w = min(scalar - slice1w, slice1w)
if cc >= 3:
output = np.zeros((height, width, cc), dtype=np.uint8)
else:
output = np.zeros((height, width), dtype=np.uint8)
if axis == 0:
output[:slice1, :] = img[:slice1, :]
output[slice1:slice1 + slice2w, :] = flip[slice1w:slice1w + slice2w, :]
else:
output[:, :slice1] = img[:, :slice1]
output[:, slice1:slice1 + slice2w] = flip[:, slice1w:slice1w + slice2w]
return output
if mode in [EnumMirrorMode.X, EnumMirrorMode.FLIP_X, EnumMirrorMode.XY, EnumMirrorMode.FLIP_XY, EnumMirrorMode.X_FLIP_Y, EnumMirrorMode.FLIP_X_FLIP_Y]:
reverse = mode in [EnumMirrorMode.FLIP_X, EnumMirrorMode.FLIP_XY, EnumMirrorMode.FLIP_X_FLIP_Y]
image = mirror(image, 1, reverse)
if mode not in [EnumMirrorMode.NONE, EnumMirrorMode.X, EnumMirrorMode.FLIP_X]:
reverse = mode in [EnumMirrorMode.FLIP_Y, EnumMirrorMode.FLIP_X_FLIP_Y, EnumMirrorMode.X_FLIP_Y]
image = mirror(image, 0, reverse)
return image
def image_pixelate(image: TYPE_IMAGE, amount:float=1.)-> TYPE_IMAGE:
h, w = image.shape[:2]
amount = max(0, min(1, amount))
block_size_h = max(1, (h * amount))
block_size_w = max(1, (w * amount))
num_blocks_h = int(np.ceil(h / block_size_h))
num_blocks_w = int(np.ceil(w / block_size_w))
block_size_h = h // num_blocks_h
block_size_w = w // num_blocks_w
pixelated_image = image.copy()
for i in range(num_blocks_h):
for j in range(num_blocks_w):
# Calculate block boundaries
y_start = i * block_size_h
y_end = min((i + 1) * block_size_h, h)
x_start = j * block_size_w
x_end = min((j + 1) * block_size_w, w)
# Average color values within the block
block_average = np.mean(image[y_start:y_end, x_start:x_end], axis=(0, 1))
# Fill the block with the average color
pixelated_image[y_start:y_end, x_start:x_end] = block_average
return pixelated_image.astype(np.uint8)
def image_posterize(image: TYPE_IMAGE, levels:int=256) -> TYPE_IMAGE:
divisor = 256 / max(2, min(256, levels))
return (np.floor(image / divisor) * int(divisor)).astype(np.uint8)
def image_quantize(image:TYPE_IMAGE, levels:int=256, iterations:int=10,
epsilon:float=0.2) -> TYPE_IMAGE:
levels = int(max(2, min(256, levels)))
pixels = np.float32(image)
criteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, iterations, epsilon)
_, labels, centers = cv2.kmeans(pixels, levels, None, criteria, 5, cv2.KMEANS_RANDOM_CENTERS)
centers = np.uint8(centers)
return centers[labels.flatten()].reshape(image.shape)
def image_rotate(image: TYPE_IMAGE, angle: float, center:TYPE_fCOORD2D=(0.5, 0.5),
edge:EnumEdge=EnumEdge.CLIP) -> TYPE_IMAGE:
@@ -185,6 +301,52 @@ def image_scale(image: TYPE_IMAGE, scale:TYPE_fCOORD2D=(1.0, 1.0),
image = image_crop_center(image, w, h)
return image
def image_scalefit(image: TYPE_IMAGE, width: int, height:int,
mode:EnumScaleMode=EnumScaleMode.MATTE,
sample:EnumInterpolation=EnumInterpolation.LANCZOS4,
matte:TYPE_PIXEL=(0,0,0,0)) -> TYPE_IMAGE:
match mode:
case EnumScaleMode.MATTE:
image = image_matte(image, matte, width, height)
case EnumScaleMode.ASPECT:
h, w = image.shape[:2]
ratio = max(width, height) / max(w, h)
image = cv2.resize(image, None, fx=ratio, fy=ratio, interpolation=sample.value)
case EnumScaleMode.ASPECT_SHORT:
h, w = image.shape[:2]
ratio = min(width, height) / min(w, h)
image = cv2.resize(image, None, fx=ratio, fy=ratio, interpolation=sample.value)
case EnumScaleMode.CROP:
image = image_crop_center(image, width, height)
matte = (*matte[:3], 0)
image = image_matte(image, matte, width, height)
case EnumScaleMode.FIT:
image = cv2.resize(image, (width, height), interpolation=sample.value)
if image.ndim == 2:
image = np.expand_dims(image, -1)
return image
def image_sharpen(image:TYPE_IMAGE, kernel_size=None, sigma:float=1.0,
amount:float=1.0, threshold:float=0) -> TYPE_IMAGE:
"""Return a sharpened version of the image, using an unsharp mask."""
kernel_size = (kernel_size, kernel_size) if kernel_size else (5, 5)
blurred = cv2.GaussianBlur(image, kernel_size, sigma)
sharpened = float(amount + 1) * image - float(amount) * blurred
sharpened = np.maximum(sharpened, np.zeros(sharpened.shape))
sharpened = np.minimum(sharpened, 255 * np.ones(sharpened.shape))
sharpened = sharpened.round().astype(np.uint8)
if threshold > 0:
low_contrast_mask = np.absolute(image - blurred) < threshold
np.copyto(sharpened, image, where=low_contrast_mask)
return sharpened
def image_translate(image: TYPE_IMAGE, offset: TYPE_fCOORD2D=(0.0, 0.0),
edge: EnumEdge=EnumEdge.CLIP, border_value:int=0) -> TYPE_IMAGE:
"""
+125
View File
@@ -0,0 +1,125 @@
"""
Jovimetrix - http://www.github.com/amorano/jovimetrix
Channel Ops
"""
from enum import Enum
from typing import List
import numpy as np
from loguru import logger
from Jovimetrix.sup.image import MIN_IMAGE_SIZE, TYPE_IMAGE, TYPE_PIXEL, \
EnumImageType, pixel_eval
# =============================================================================
# === ENUMERATION ===
# =============================================================================
class EnumPixelSwizzle(Enum):
RED_A = 20
GREEN_A = 10
BLUE_A = 0
ALPHA_A = 30
RED_B = 21
GREEN_B = 11
BLUE_B = 1
ALPHA_B = 31
CONSTANT = 50
# =============================================================================
# === CHANNEL ===
# =============================================================================
def channel_add(image:TYPE_IMAGE, color:TYPE_PIXEL=255) -> TYPE_IMAGE:
"""
This function adds a new channel with a solid color to an image.
:param image: The `image` parameter is expected to be an image represented as a
NumPy array. The function assumes that the image has a shape attribute that
returns a tuple representing the dimensions of the image (height, width, and
channels if it's a color image)
:type image: TYPE_IMAGE
:param color: The `color` parameter in the `channel_add` function represents the
color value that will be added as a new channel to the input image. The default
value for `color` is 255, which is typically a white color in grayscale images,
defaults to 255
:type color: TYPE_PIXEL (optional)
:return: The function `channel_add` returns a new image with an additional
channel appended to the original image. The new channel has a solid color
specified by the `color` parameter.
"""
h, w = image.shape[:2]
color = pixel_eval(color, EnumImageType.GRAYSCALE)
new = channel_solid(w, h, color, EnumImageType.GRAYSCALE)
return np.concatenate([image, new], axis=-1)
def channel_solid(width:int=MIN_IMAGE_SIZE, height:int=MIN_IMAGE_SIZE, color:TYPE_PIXEL=(0, 0, 0, 255),
chan:EnumImageType=EnumImageType.BGR) -> TYPE_IMAGE:
if chan == EnumImageType.GRAYSCALE:
color = pixel_eval(color, EnumImageType.GRAYSCALE)
what = np.full((height, width, 1), color, dtype=np.uint8)
return what
if not type(color) in [list, set, tuple]:
color = [color]
color += (0,) * (3 - len(color))
if chan in [EnumImageType.BGR, EnumImageType.RGB]:
if chan == EnumImageType.RGB:
color = color[2::-1]
return np.full((height, width, 3), color[:3], dtype=np.uint8)
if len(color) < 4:
color += (255,)
if chan == EnumImageType.RGBA:
color = color[2::-1]
return np.full((height, width, 4), color, dtype=np.uint8)
def channel_merge(channels: List[TYPE_IMAGE]) -> TYPE_IMAGE:
max_height = max(ch.shape[0] for ch in channels if ch is not None)
max_width = max(ch.shape[1] for ch in channels if ch is not None)
num_channels = len(channels)
dtype = channels[0].dtype
output = np.zeros((max_height, max_width, num_channels), dtype=dtype)
for i, channel in enumerate(channels):
if channel is None:
continue
h, w = channel.shape[:2]
if channel.ndim > 2:
channel = channel[..., 0]
pad_top = (max_height - h) // 2
pad_bottom = max_height - h - pad_top
pad_left = (max_width - w) // 2
pad_right = max_width - w - pad_left
padded_channel = np.pad(channel, ((pad_top, pad_bottom), (pad_left, pad_right)),
mode='constant', constant_values=0)
output[..., i] = padded_channel
if num_channels == 1:
output = output[..., 0]
return output
def channel_swap(imageA:TYPE_IMAGE, swap_ot:EnumPixelSwizzle,
imageB:TYPE_IMAGE, swap_in:EnumPixelSwizzle) -> TYPE_IMAGE:
index_out = int(swap_ot.value / 10)
cc_out = imageA.shape[2] if imageA.ndim == 3 else 1
# swap channel is out of range of image size
if index_out > cc_out:
return imageA
index_in = int(swap_in.value / 10)
cc_in = imageB.shape[2] if imageB.ndim == 3 else 1
if index_in > cc_in:
return imageA
imageA[:,:,index_out] = imageB[:,:,index_in]
return imageA
+154 -4
View File
@@ -3,6 +3,7 @@ Jovimetrix - http://www.github.com/amorano/jovimetrix
Image Color Support
"""
from enum import Enum
from typing import Tuple
import cv2
@@ -12,12 +13,142 @@ from daltonlens import simulate
from skimage import exposure
from blendmodes.blend import BlendType
from Jovimetrix.sup.image import TYPE_IMAGE, TYPE_PIXEL, EnumCBDeficiency, \
EnumCBSimulator, EnumColorTheory, bgr2hsv, hsv2bgr, image_blend, \
image_convert, image_mask, image_mask_add, pixel_hsv_adjust
from Jovimetrix.sup.image import TYPE_IMAGE, TYPE_PIXEL, bgr2hsv, hsv2bgr, \
image_blend, image_convert, image_grayscale, image_mask, image_mask_add
# =============================================================================
# === COLOR FUNCTIONS ===
# === ENUMERATION ===
# =============================================================================
class EnumColorMap(Enum):
AUTUMN = cv2.COLORMAP_AUTUMN
BONE = cv2.COLORMAP_BONE
JET = cv2.COLORMAP_JET
WINTER = cv2.COLORMAP_WINTER
RAINBOW = cv2.COLORMAP_RAINBOW
OCEAN = cv2.COLORMAP_OCEAN
SUMMER = cv2.COLORMAP_SUMMER
SPRING = cv2.COLORMAP_SPRING
COOL = cv2.COLORMAP_COOL
HSV = cv2.COLORMAP_HSV
PINK = cv2.COLORMAP_PINK
HOT = cv2.COLORMAP_HOT
PARULA = cv2.COLORMAP_PARULA
MAGMA = cv2.COLORMAP_MAGMA
INFERNO = cv2.COLORMAP_INFERNO
PLASMA = cv2.COLORMAP_PLASMA
VIRIDIS = cv2.COLORMAP_VIRIDIS
CIVIDIS = cv2.COLORMAP_CIVIDIS
TWILIGHT = cv2.COLORMAP_TWILIGHT
TWILIGHT_SHIFTED = cv2.COLORMAP_TWILIGHT_SHIFTED
TURBO = cv2.COLORMAP_TURBO
DEEPGREEN = cv2.COLORMAP_DEEPGREEN
class EnumColorTheory(Enum):
COMPLIMENTARY = 0
MONOCHROMATIC = 1
SPLIT_COMPLIMENTARY = 2
ANALOGOUS = 3
TRIADIC = 4
# TETRADIC = 5
SQUARE = 6
COMPOUND = 8
# DOUBLE_COMPLIMENTARY = 9
CUSTOM_TETRAD = 9
class EnumCBDeficiency(Enum):
PROTAN = simulate.Deficiency.PROTAN
DEUTAN = simulate.Deficiency.DEUTAN
TRITAN = simulate.Deficiency.TRITAN
class EnumCBSimulator(Enum):
AUTOSELECT = 0
BRETTEL1997 = 1
COBLISV1 = 2
COBLISV2 = 3
MACHADO2009 = 4
VIENOT1999 = 5
VISCHECK = 6
# ==============================================================================
# === COLOR SPACE CONVERSION ===
# ==============================================================================
def gamma2linear(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Gamma correction for old PCs/CRT monitors"""
return np.power(image, 2.2)
def linear2gamma(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Inverse gamma correction for old PCs/CRT monitors"""
return np.power(np.clip(image, 0., 1.), 1.0 / 2.2)
def sRGB2Linear(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Convert sRGB to linearRGB, removing the gamma correction.
Works for grayscale, RGB, or RGBA images.
"""
image = image.astype(float) / 255.0
# If the image has an alpha channel, separate it
if image.shape[-1] == 4:
rgb = image[..., :3]
alpha = image[..., 3]
else:
rgb = image
alpha = None
gamma = ((rgb + 0.055) / 1.055) ** 2.4
scale = rgb / 12.92
rgb = np.where(rgb > 0.04045, gamma, scale)
# Recombine the alpha channel if it exists
if alpha is not None:
image = np.concatenate((rgb, alpha[..., np.newaxis]), axis=-1)
else:
image = rgb
return (image * 255).astype(np.uint8)
def linear2sRGB(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Convert linearRGB to sRGB, applying the gamma correction.
Works for grayscale, RGB, or RGBA images.
"""
image = image.astype(float) / 255.0
# If the image has an alpha channel, separate it
if image.shape[-1] == 4:
rgb = image[..., :3]
alpha = image[..., 3]
else:
rgb = image
alpha = None
higher = 1.055 * np.power(rgb, 1.0 / 2.4) - 0.055
lower = rgb * 12.92
rgb = np.where(rgb > 0.0031308, higher, lower)
# Recombine the alpha channel if it exists
if alpha is not None:
image = np.concatenate((rgb, alpha[..., np.newaxis]), axis=-1)
else:
image = rgb
return np.clip(image * 255.0, 0, 255).astype(np.uint8)
# ==============================================================================
# === PIXEL ===
# ==============================================================================
def pixel_hsv_adjust(color:TYPE_PIXEL, hue:int=0, saturation:int=0, value:int=0,
mod_color:bool=True, mod_sat:bool=False,
mod_value:bool=False) -> TYPE_PIXEL:
"""Adjust an HSV type pixel.
OpenCV uses... H: 0-179, S: 0-255, V: 0-255"""
hsv = [0, 0, 0]
hsv[0] = (color[0] + hue) % 180 if mod_color else np.clip(color[0] + hue, 0, 180)
hsv[1] = (color[1] + saturation) % 255 if mod_sat else np.clip(color[1] + saturation, 0, 255)
hsv[2] = (color[2] + value) % 255 if mod_value else np.clip(color[2] + value, 0, 255)
return hsv
# =============================================================================
# === COLOR MATCH ===
# =============================================================================
@cuda.jit
@@ -186,6 +317,10 @@ def color_mean(image: TYPE_IMAGE) -> TYPE_IMAGE:
int(np.mean(image[:,:,2])) ]
return color
# ==============================================================================
# === COLOR ANALYSIS ===
# ==============================================================================
def color_theory_complementary(color: TYPE_PIXEL) -> TYPE_PIXEL:
color = bgr2hsv(color)
color_a = pixel_hsv_adjust(color, 90, 0, 0)
@@ -284,3 +419,18 @@ def color_theory(image: TYPE_IMAGE, custom:int=0, scheme: EnumColorTheory=EnumCo
np.full((h, w, 3), c, dtype=np.uint8),
np.full((h, w, 3), d, dtype=np.uint8),
)
#
#
#
# Adapted from WAS Suite -- gradient_map
# https://github.com/WASasquatch/was-node-suite-comfyui
def image_gradient_map(image:TYPE_IMAGE, gradient_map:TYPE_IMAGE, reverse:bool=False) -> TYPE_IMAGE:
if reverse:
gradient_map = gradient_map[:,:,::-1]
grey = image_grayscale(image)
cmap = image_convert(gradient_map, 3)
cmap = cv2.resize(cmap, (256, 256))
cmap = cmap[0,:,:].reshape((256, 1, 3)).astype(np.uint8)
return cv2.applyColorMap(grey, cmap)
+234 -195
View File
@@ -3,210 +3,176 @@ Jovimetrix - http://www.github.com/amorano/jovimetrix
Image Composition Operation Support
"""
from enum import Enum
import sys
from typing import List, Tuple
import cv2
import numpy as np
from blendmodes.blend import BlendType
from loguru import logger
from Jovimetrix.sup.image import TYPE_IMAGE
from Jovimetrix.sup.image import TYPE_IMAGE, TYPE_PIXEL, EnumInterpolation, \
EnumScaleMode, bgr2image, image2bgr, image_blend, image_convert, \
image_mask, image_matte, image_minmax
from Jovimetrix.sup.image.misc import image_detect, image_grayscale
from Jovimetrix.sup.image.adjust import image_scalefit
# =============================================================================
# ==============================================================================
# === ENUMERATION ===
# ==============================================================================
class EnumAdjustOP(Enum):
BLUR = 0
STACK_BLUR = 1
GAUSSIAN_BLUR = 2
MEDIAN_BLUR = 3
SHARPEN = 10
EMBOSS = 20
INVERT = 25
# MEAN = 30 -- in UNARY
# ADAPTIVE_HISTOGRAM = 35
HSV = 30
LEVELS = 35
EQUALIZE = 40
PIXELATE = 50
QUANTIZE = 55
POSTERIZE = 60
FIND_EDGES = 80
OUTLINE = 70
DILATE = 71
ERODE = 72
OPEN = 73
CLOSE = 74
class EnumBlendType(Enum):
"""Rename the blend type names."""
NORMAL = BlendType.NORMAL
ADDITIVE = BlendType.ADDITIVE
NEGATION = BlendType.NEGATION
DIFFERENCE = BlendType.DIFFERENCE
MULTIPLY = BlendType.MULTIPLY
DIVIDE = BlendType.DIVIDE
LIGHTEN = BlendType.LIGHTEN
DARKEN = BlendType.DARKEN
SCREEN = BlendType.SCREEN
BURN = BlendType.COLOURBURN
DODGE = BlendType.COLOURDODGE
OVERLAY = BlendType.OVERLAY
HUE = BlendType.HUE
SATURATION = BlendType.SATURATION
LUMINOSITY = BlendType.LUMINOSITY
COLOR = BlendType.COLOUR
SOFT = BlendType.SOFTLIGHT
HARD = BlendType.HARDLIGHT
PIN = BlendType.PINLIGHT
VIVID = BlendType.VIVIDLIGHT
EXCLUSION = BlendType.EXCLUSION
REFLECT = BlendType.REFLECT
GLOW = BlendType.GLOW
XOR = BlendType.XOR
EXTRACT = BlendType.GRAINEXTRACT
MERGE = BlendType.GRAINMERGE
DESTIN = BlendType.DESTIN
DESTOUT = BlendType.DESTOUT
SRCATOP = BlendType.SRCATOP
DESTATOP = BlendType.DESTATOP
class EnumImageBySize(Enum):
LARGEST = 10
SMALLEST = 20
WIDTH_MIN = 30
WIDTH_MAX = 40
HEIGHT_MIN = 50
HEIGHT_MAX = 60
class EnumOrientation(Enum):
HORIZONTAL = 0
VERTICAL = 1
GRID = 2
# ==============================================================================
# === PIXEL ===
# ==============================================================================
def pixel_convert(color:TYPE_PIXEL, size:int=4, alpha:int=255) -> TYPE_PIXEL:
"""Convert X channel pixel into Y channel pixel."""
if (cc := len(color)) == size:
return color
if size > 2:
color += (0,) * (3 - cc)
if size == 4:
color += (alpha,)
return color
return color[0]
# ==============================================================================
# === IMAGE ===
# =============================================================================
# ==============================================================================
"""
These are core functions that most of the support image libraries require.
"""
def image_crop_head(image: TYPE_IMAGE) -> TYPE_IMAGE:
def image_flatten(image: List[TYPE_IMAGE], width:int=None, height:int=None,
mode=EnumScaleMode.MATTE,
sample:EnumInterpolation=EnumInterpolation.LANCZOS4) -> TYPE_IMAGE:
if mode == EnumScaleMode.MATTE:
width, height, _, _ = image_minmax(image)[1:]
else:
h, w = image[0].shape[:2]
width = width or w
height = height or h
current = np.zeros((height, width, 4), dtype=np.uint8)
for x in image:
if mode != EnumScaleMode.MATTE:
x = image_scalefit(x, width, height, mode, sample)
x = image_matte(x, (0,0,0,0), width, height)
x = image_scalefit(x, width, height, EnumScaleMode.CROP, sample)
x = image_convert(x, 4)
#@TODO: ADD VARIOUS COMP OPS?
current = cv2.add(current, x)
return current
def image_flatten_mask(image:TYPE_IMAGE, matte:Tuple=(0,0,0,255)) -> Tuple[TYPE_IMAGE, TYPE_IMAGE|None]:
"""Flatten the image with its own alpha channel, if any."""
mask = image_mask(image)
return image_blend(image, image, mask), mask
def image_levels(image: np.ndarray, black_point:int=0, white_point=255,
mid_point=128, gamma=1.0) -> np.ndarray:
"""
Given a file path or np.ndarray image with a face,
returns cropped np.ndarray around the largest detected
face.
Adjusts the levels of an image including black, white, midpoints, and gamma correction.
Parameters
----------
- `path_or_array` : {`str`, `np.ndarray`}
* The filepath or numpy array of the image.
Args:
image (numpy.ndarray): Input image tensor in RGB(A) format.
black_point (int): The black point to adjust shadows. Default is 0.
white_point (int): The white point to adjust highlights. Default is 255.
mid_point (int): The mid point for mid-tone adjustment. Default is 128.
gamma (float): Gamma correction value. Default is 1.0.
Returns
-------
- `image` : {`np.ndarray`, `None`}
* A cropped numpy array if face detected, else None.
Returns:
numpy.ndarray: Adjusted image tensor.
"""
MIN_FACE = 8
image, alpha, cc = image2bgr(image)
gray = image_grayscale(image)
h, w = image.shape[:2]
minface = int(np.sqrt(h**2 + w**2) / MIN_FACE)
# Convert points and gamma to float32 for calculations
black = np.array([black_point] * 3, dtype=np.float32)
white = np.array([white_point] * 3, dtype=np.float32)
mid = np.array([mid_point] * 3, dtype=np.float32)
inGamma = np.array([gamma] * 3, dtype=np.float32)
outBlack = np.array([0, 0, 0], dtype=np.float32)
outWhite = np.array([255, 255, 255], dtype=np.float32)
'''
# Create the haar cascade
face_cascade = cv2.CascadeClassifier(self.casc_path)
# ====== Detect faces in the image ======
faces = face_cascade.detectMultiScale(
gray,
scaleFactor=1.1,
minNeighbors=5,
minSize=(minface, minface),
flags=cv2.CASCADE_FIND_BIGGEST_OBJECT | cv2.CASCADE_DO_ROUGH_SEARCH,
)
# Handle no faces
if len(faces) == 0:
return None
# Make padding from biggest face found
x, y, w, h = faces[-1]
pos = self._crop_positions(
img_height,
img_width,
x,
y,
w,
h,
)
# ====== Actual cropping ======
image = image[pos[0] : pos[1], pos[2] : pos[3]]
# Resize
if self.resize:
with Image.fromarray(image) as img:
image = np.array(img.resize((self.width, self.height)))
# Underexposition fix
if self.gamma:
image = check_underexposed(image, gray)
return bgr_to_rbg(image)
def _determine_safe_zoom(self, imgh, imgw, x, y, w, h):
"""
Determines the safest zoom level with which to add margins
around the detected face. Tries to honor `self.face_percent`
when possible.
Parameters:
-----------
imgh: int
Height (px) of the image to be cropped
imgw: int
Width (px) of the image to be cropped
x: int
Leftmost coordinates of the detected face
y: int
Bottom-most coordinates of the detected face
w: int
Width of the detected face
h: int
Height of the detected face
Diagram:
--------
i / j := zoom / 100
+
h1 | h2
+---------|---------+
| MAR|GIN |
| (x+w, y+h)|
| +-----|-----+ |
| | FA|CE | |
| | | | |
| ├──i──┤ | |
| | cen|ter | |
| | | | |
| +-----|-----+ |
| (x, y)| |
| | |
+---------|---------+
├────j────┤
+
"""
# Find out what zoom factor to use given self.aspect_ratio
corners = itertools.product((x, x + w), (y, y + h))
center = np.array([x + int(w / 2), y + int(h / 2)])
i = np.array(
[(0, 0), (0, imgh), (imgw, imgh), (imgw, 0), (0, 0)]
) # image_corners
image_sides = [(i[n], i[n + 1]) for n in range(4)]
corner_ratios = [self.face_percent] # Hopefully we use this one
for c in corners:
corner_vector = np.array([center, c])
a = distance(*corner_vector)
intersects = list(intersect(corner_vector, side) for side in image_sides)
for pt in intersects:
if (pt >= 0).all() and (pt <= i[2]).all(): # if intersect within image
dist_to_pt = distance(center, pt)
corner_ratios.append(100 * a / dist_to_pt)
return max(corner_ratios)
def _crop_positions(
self,
imgh,
imgw,
x,
y,
w,
h,
):
"""
Retuns the coordinates of the crop position centered
around the detected face with extra margins. Tries to
honor `self.face_percent` if possible, else uses the
largest margins that comply with required aspect ratio
given by `self.height` and `self.width`.
Parameters:
-----------
imgh: int
Height (px) of the image to be cropped
imgw: int
Width (px) of the image to be cropped
x: int
Leftmost coordinates of the detected face
y: int
Bottom-most coordinates of the detected face
w: int
Width of the detected face
h: int
Height of the detected face
"""
zoom = self._determine_safe_zoom(imgh, imgw, x, y, w, h)
# Adjust output height based on percent
if self.height >= self.width:
height_crop = h * 100.0 / zoom
width_crop = self.aspect_ratio * float(height_crop)
else:
width_crop = w * 100.0 / zoom
height_crop = float(width_crop) / self.aspect_ratio
# Calculate padding by centering face
xpad = (width_crop - w) / 2
ypad = (height_crop - h) / 2
# Calc. positions of crop
h1 = x - xpad
h2 = x + w + xpad
v1 = y - ypad
v2 = y + h + ypad
return [int(v1), int(v2), int(h1), int(h2)]
'''
def image_histogram_statistics(histogram:np.ndarray, L=256)-> TYPE_IMAGE:
sumPixels = np.sum(histogram)
normalizedHistogram = histogram/sumPixels
mean = 0
for i in range(L):
mean += i * normalizedHistogram[i]
variance = 0
for i in range(L):
variance += (i-mean)**2 * normalizedHistogram[i]
std = np.sqrt(variance)
return mean, variance, std
# Apply levels adjustment
image = np.clip((image - black) / (white - black), 0, 1)
image = (image - mid) / (1.0 - mid)
image = (image ** (1 / inGamma)) * (outWhite - outBlack) + outBlack
image = np.clip(image, 0, 255).astype(np.uint8)
return bgr2image(image, alpha, cc == 1)
def image_mask_binary(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""
@@ -244,10 +210,83 @@ def image_mask_binary(image: TYPE_IMAGE) -> TYPE_IMAGE:
mask = np.expand_dims(mask, -1)
return mask.astype(np.uint8)
def image_recenter(image: TYPE_IMAGE) -> TYPE_IMAGE:
cropped_image = image_detect(image)[0]
new_image = np.zeros(image.shape, dtype=np.uint8)
paste_x = (new_image.shape[1] - cropped_image.shape[1]) // 2
paste_y = (new_image.shape[0] - cropped_image.shape[0]) // 2
new_image[paste_y:paste_y+cropped_image.shape[0], paste_x:paste_x+cropped_image.shape[1]] = cropped_image
return new_image
def image_by_size(image_list: List[TYPE_IMAGE],
enumSize: EnumImageBySize=EnumImageBySize.LARGEST) -> Tuple[TYPE_IMAGE, int, int]:
img = None
mega, width, height = 0, 0, 0
if enumSize in [EnumImageBySize.SMALLEST, EnumImageBySize.WIDTH_MIN, EnumImageBySize.HEIGHT_MIN]:
mega, width, height = sys.maxsize, sys.maxsize, sys.maxsize
for i in image_list:
h, w = i.shape[:2]
match enumSize:
case EnumImageBySize.LARGEST:
if (new_mega := w * h) > mega:
mega = new_mega
img = i
width = max(width, w)
height = max(height, h)
case EnumImageBySize.SMALLEST:
if (new_mega := w * h) < mega:
mega = new_mega
img = i
width = min(width, w)
height = min(height, h)
case EnumImageBySize.WIDTH_MIN:
if w < width:
width = w
img = i
case EnumImageBySize.WIDTH_MAX:
if w > width:
width = w
img = i
case EnumImageBySize.HEIGHT_MIN:
if h < height:
height = h
img = i
case EnumImageBySize.HEIGHT_MAX:
if h > height:
height = h
img = i
return img, width, height
def image_stack(image_list: List[TYPE_IMAGE],
axis:EnumOrientation=EnumOrientation.HORIZONTAL,
stride:int=0, matte:TYPE_PIXEL=(0,0,0,255)) -> TYPE_IMAGE:
_, width, height = image_by_size(image_list)
images = [image_matte(image_convert(i, 4), matte, width, height) for i in image_list]
count = len(images)
matte = pixel_convert(matte, 4)
match axis:
case EnumOrientation.GRID:
if stride < 1:
stride = np.ceil(np.sqrt(count))
stride = int(stride)
stride = min(stride, count)
stride = max(stride, 1)
rows = []
for i in range(0, count, stride):
row = images[i:i + stride]
row_stacked = np.hstack(row)
rows.append(row_stacked)
height, width = images[0].shape[:2]
overhang = count % stride
if overhang != 0:
overhang = stride - overhang
size = (height, overhang * width, 4)
filler = np.full(size, matte, dtype=np.uint8)
rows[-1] = np.hstack([rows[-1], filler])
image = np.vstack(rows)
case EnumOrientation.HORIZONTAL:
image = np.hstack(images)
case EnumOrientation.VERTICAL:
image = np.vstack(images)
return image
+248
View File
@@ -0,0 +1,248 @@
"""
Jovimetrix - http://www.github.com/amorano/jovimetrix
Coordinates and Mapping
"""
from typing import Any, List, Tuple
import cv2
import numpy as np
from loguru import logger
from Jovimetrix.sup.image import TAU, TYPE_IMAGE, TYPE_fCOORD2D, image_lerp, \
image_normalize
from Jovimetrix.sup.image.color import image_grayscale
# =============================================================================
# === IMAGE ===
# =============================================================================
def image_mirror_mandela(imageA: np.ndarray, imageB: np.ndarray) -> Tuple[np.ndarray, ...]:
"""Merge 4 flipped copies of input images to make them wrap.
Output is twice bigger in both dimensions."""
top = np.hstack([imageA, -np.flip(imageA, axis=1)])
bottom = np.hstack([np.flip(imageA, axis=0), -np.flip(imageA)])
imageA = np.vstack([top, bottom])
top = np.hstack([imageB, np.flip(imageB, axis=1)])
bottom = np.hstack([-np.flip(imageB, axis=0), -np.flip(imageB)])
imageB = np.vstack([top, bottom])
return imageA, imageB
# =============================================================================
# === COORDINATES ===
# =============================================================================
def coord_cart2polar(x: float, y: float) -> TYPE_fCOORD2D:
r = np.sqrt(x**2 + y**2)
theta = np.arctan2(y, x)
return r, theta
def coord_polar2cart(r: float, theta: float) -> TYPE_fCOORD2D:
x = r * np.cos(theta)
y = r * np.sin(theta)
return x, y
def coord_default(width:int, height:int, origin:TYPE_fCOORD2D=None) -> TYPE_fCOORD2D:
"""Creates x & y coords for the indicies in a numpy array "data".
"origin" defaults to the center of the image. Specify origin=(0,0)
to set the origin to the lower left corner of the image."""
if origin is None:
origin_x, origin_y = width // 2, height // 2
else:
origin_x, origin_y = origin
x, y = np.meshgrid(np.arange(width), np.arange(height))
x -= origin_x
y -= origin_y
return x, y
def coord_fisheye(width: int, height: int, distortion: float) -> Tuple[TYPE_IMAGE, TYPE_IMAGE]:
map_x, map_y = np.meshgrid(np.linspace(0., 1., width), np.linspace(0., 1., height))
# normalized
xnd, ynd = (2 * map_x - 1), (2 * map_y - 1)
rd = np.sqrt(xnd**2 + ynd**2)
# fish-eye distortion
condition = (dist := 1 - distortion * (rd**2)) == 0
xdu, ydu = np.where(condition, xnd, xnd / dist), np.where(condition, ynd, ynd / dist)
xu, yu = ((xdu + 1) * width) / 2, ((ydu + 1) * height) / 2
return xu.astype(np.float32), yu.astype(np.float32)
def coord_perspective(width: int, height: int, pts: List[TYPE_fCOORD2D]) -> TYPE_IMAGE:
object_pts = np.float32([[0, 0], [width, 0], [width, height], [0, height]])
pts = np.float32(pts)
pts = np.column_stack([pts[:, 0], pts[:, 1]])
return cv2.getPerspectiveTransform(object_pts, pts)
def coord_sphere(width: int, height: int, radius: float) -> Tuple[TYPE_IMAGE, TYPE_IMAGE]:
theta, phi = np.meshgrid(np.linspace(0, TAU, width), np.linspace(0, np.pi, height))
x = radius * np.sin(phi) * np.cos(theta)
y = radius * np.sin(phi) * np.sin(theta)
# z = radius * np.cos(phi)
x_image = (x + 1) * (width - 1) / 2
y_image = (y + 1) * (height - 1) / 2
return x_image.astype(np.float32), y_image.astype(np.float32)
# =============================================================================
# === MAPPING ===
# =============================================================================
def remap_fisheye(image: TYPE_IMAGE, distort: float) -> TYPE_IMAGE:
cc = image.shape[2] if image.ndim == 3 else 1
height, width = image.shape[:2]
if cc == 1:
image = cv2.cvtColor(image, cv2.COLOR_GRAY2BGR)
map_x, map_y = coord_fisheye(width, height, distort)
image = cv2.remap(image, map_x, map_y, interpolation=cv2.INTER_LINEAR, borderMode=cv2.BORDER_CONSTANT)
#if cc == 1:
# image = image[..., 0]
return image
def remap_perspective(image: TYPE_IMAGE, pts: list) -> TYPE_IMAGE:
cc = image.shape[2] if image.ndim == 3 else 1
height, width = image.shape[:2]
if cc == 1:
image = cv2.cvtColor(image, cv2.COLOR_GRAY2BGR)
pts = coord_perspective(width, height, pts)
image = cv2.warpPerspective(image, pts, (width, height))
#if cc == 1:
# image = image[..., 0]
return image
def remap_polar(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Re-projects a 3D numpy array ("data") into a polar coordinate system.
"origin" is a tuple of (x0, y0) and defaults to the center of the image."""
h, w = image.shape[:2]
radius = max(w, h)
return cv2.linearPolar(image, (h // 2, w // 2), radius // 2, cv2.WARP_INVERSE_MAP)
def remap_sphere(image: TYPE_IMAGE, radius: float) -> TYPE_IMAGE:
height, width = image.shape[:2]
map_x, map_y = coord_sphere(width, height, radius)
return cv2.remap(image, map_x, map_y, interpolation=cv2.INTER_LINEAR, borderMode=cv2.BORDER_CONSTANT)
def depth_from_gradient(grad_x, grad_y):
"""Optimized Frankot-Chellappa depth-from-gradient algorithm."""
rows, cols = grad_x.shape
rows_scale = np.fft.fftfreq(rows)
cols_scale = np.fft.fftfreq(cols)
u_grid, v_grid = np.meshgrid(cols_scale, rows_scale)
grad_x_F = np.fft.fft2(grad_x)
grad_y_F = np.fft.fft2(grad_y)
denominator = u_grid**2 + v_grid**2
denominator[0, 0] = 1.0
Z_F = (-1j * u_grid * grad_x_F - 1j * v_grid * grad_y_F) / denominator
Z_F[0, 0] = 0.0
Z = np.fft.ifft2(Z_F).real
Z -= np.min(Z)
Z /= np.max(Z)
return Z
def height_from_normal(image: TYPE_IMAGE, tile:bool=True) -> TYPE_IMAGE:
"""Computes a height map from the given normal map."""
image = np.transpose(image, (2, 0, 1))
flip_img = np.flip(image, axis=1)
grad_x, grad_y = (flip_img[0] - 0.5) * 2, (flip_img[1] - 0.5) * 2
grad_x = np.flip(grad_x, axis=0)
grad_y = np.flip(grad_y, axis=0)
if not tile:
grad_x, grad_y = image_mirror_mandela(grad_x, grad_y)
pred_img = depth_from_gradient(-grad_x, grad_y)
# re-crop
if not tile:
height, width = image.shape[1], image.shape[2]
pred_img = pred_img[:height, :width]
image = np.stack([pred_img, pred_img, pred_img])
image = np.transpose(image, (1, 2, 0))
return image
def curvature_from_normal(image: TYPE_IMAGE, blur_radius:int=2)-> TYPE_IMAGE:
"""Computes a curvature map from the given normal map."""
image = np.transpose(image, (2, 0, 1))
blur_factor = 1 / 2 ** min(8, max(2, blur_radius))
diff_kernel = np.array([-1, 0, 1])
def conv_1d(array, kernel) -> np.ndarray[Any, np.dtype[Any]]:
"""Performs row-wise 1D convolutions with repeat padding."""
k_l = len(kernel)
extended = np.pad(array, k_l // 2, mode="wrap")
return np.array([np.convolve(row, kernel, mode="valid") for row in extended[k_l//2:-k_l//2+1]])
h_conv = conv_1d(image[0], diff_kernel)
v_conv = conv_1d(-image[1].T, diff_kernel).T
edges_conv = h_conv + v_conv
# Calculate blur radius in pixels
blur_radius_px = int(np.mean(image.shape[1:3]) * blur_factor)
if blur_radius_px < 2:
# If blur radius is too small, just normalize the edge convolution
image = (edges_conv - np.min(edges_conv)) / (np.ptp(edges_conv) + 1e-10)
else:
blur_radius_px += blur_radius_px % 2 == 0
# Compute Gaussian kernel
sigma = max(1, blur_radius_px // 8)
x = np.linspace(-(blur_radius_px - 1) / 2, (blur_radius_px - 1) / 2, blur_radius_px)
g_kernel = np.exp(-0.5 * np.square(x) / np.square(sigma))
g_kernel /= np.sum(g_kernel)
# Apply Gaussian blur
h_blur = conv_1d(edges_conv, g_kernel)
v_blur = conv_1d(h_blur.T, g_kernel).T
image = (v_blur - np.min(v_blur)) / (np.ptp(v_blur) + 1e-10)
image = (image - image.min()) / (image.max() - image.min()) * 255
return image.astype(np.uint8)
def roughness_from_normal(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Roughness from a normal map."""
up_vector = np.array([0, 0, 1])
image = 1 - np.dot(image, up_vector)
image = (image - image.min()) / (image.max() - image.min())
image = (255 * image).astype(np.uint8)
return image_grayscale(image)
def roughness_from_albedo(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Roughness from an albedo map."""
kernel_size = 3
image = cv2.Laplacian(image, cv2.CV_64F, ksize=kernel_size)
image = (image - image.min()) / (image.max() - image.min())
image = (255 * image).astype(np.uint8)
return image_grayscale(image)
def roughness_from_albedo_normal(albedo: TYPE_IMAGE, normal: TYPE_IMAGE,
blur:int=2, blend:float=0.5, iterations:int=3) -> TYPE_IMAGE:
normal = roughness_from_normal(normal)
normal = image_normalize(normal)
albedo = roughness_from_albedo(albedo)
albedo = image_normalize(albedo)
rough = image_lerp(normal, albedo, alpha=blend)
rough = image_normalize(rough)
image = image_lerp(normal, rough, alpha=blend)
iterations = min(16, max(2, iterations))
blur += (blur % 2 == 0)
step = 1 / 2 ** iterations
for i in range(iterations):
image = cv2.add(normal * step, image * step)
image = cv2.GaussianBlur(image, (blur + i * 2, blur + i * 2), 3 * i)
inverted = 255 - image_normalize(image)
inverted = cv2.subtract(inverted, albedo) * 0.5
inverted = cv2.GaussianBlur(inverted, (blur, blur), blur)
inverted = image_normalize(inverted)
image = cv2.add(image * 0.5, inverted * 0.5)
for i in range(iterations):
image = cv2.GaussianBlur(image, (blur, blur), blur)
image = cv2.add(image * 0.5, inverted * 0.5)
for i in range(iterations):
image = cv2.GaussianBlur(image, (blur, blur), blur)
image = image_normalize(image)
return image
+32 -656
View File
@@ -4,25 +4,46 @@ Extras Support
"""
import math
import sys
from typing import Any, List, Tuple
from enum import Enum
from typing import Any, Tuple
import cv2
import numpy as np
from numba import jit
from scipy import ndimage
from skimage.metrics import structural_similarity as ssim
from PIL import Image, ImageDraw, ImageChops
from PIL import Image, ImageDraw
from loguru import logger
from Jovimetrix.sup.util import grid_make
from Jovimetrix.sup.image import TYPE_IMAGE, TYPE_PIXEL, \
TYPE_iRGB, bgr2image, image2bgr, image_convert, pil2cv
from Jovimetrix.sup.image import TAU, TYPE_IMAGE, TYPE_PIXEL, TYPE_fCOORD2D, \
TYPE_iRGB, EnumImageBySize, EnumMirrorMode, EnumOrientation, \
EnumThreshold, EnumThresholdAdapt, bgr2image, channel_add, cv2pil, \
image2bgr, image_grayscale, image_matte, image_normalize, pil2cv, \
pixel_convert, image_convert
# =============================================================================
# === ENUMERATION ===
# =============================================================================
class EnumProjection(Enum):
NORMAL = 0
POLAR = 5
SPHERICAL = 10
FISHEYE = 15
PERSPECTIVE = 20
class EnumShapes(Enum):
CIRCLE = 0
SQUARE = 1
ELLIPSE = 2
RECTANGLE = 3
POLYGON = 4
class EnumThreshold(Enum):
BINARY = cv2.THRESH_BINARY
TRUNC = cv2.THRESH_TRUNC
TOZERO = cv2.THRESH_TOZERO
class EnumThresholdAdapt(Enum):
ADAPT_NONE = -1
ADAPT_MEAN = cv2.ADAPTIVE_THRESH_MEAN_C
ADAPT_GAUSS = cv2.ADAPTIVE_THRESH_GAUSSIAN_C
# =============================================================================
# === EXPLICIT SHAPE FUNCTIONS ===
@@ -56,113 +77,6 @@ def shape_polygon(width: int, height: int, size: float=1., sides: int=3,
d.regular_polygon(xy, sides, fill=fill)
return image
def image_by_size(image_list: List[TYPE_IMAGE],
enumSize: EnumImageBySize=EnumImageBySize.LARGEST) -> Tuple[TYPE_IMAGE, int, int]:
img = None
mega, width, height = 0, 0, 0
if enumSize in [EnumImageBySize.SMALLEST, EnumImageBySize.WIDTH_MIN, EnumImageBySize.HEIGHT_MIN]:
mega, width, height = sys.maxsize, sys.maxsize, sys.maxsize
for i in image_list:
h, w = i.shape[:2]
match enumSize:
case EnumImageBySize.LARGEST:
if (new_mega := w * h) > mega:
mega = new_mega
img = i
width = max(width, w)
height = max(height, h)
case EnumImageBySize.SMALLEST:
if (new_mega := w * h) < mega:
mega = new_mega
img = i
width = min(width, w)
height = min(height, h)
case EnumImageBySize.WIDTH_MIN:
if w < width:
width = w
img = i
case EnumImageBySize.WIDTH_MAX:
if w > width:
width = w
img = i
case EnumImageBySize.HEIGHT_MIN:
if h < height:
height = h
img = i
case EnumImageBySize.HEIGHT_MAX:
if h > height:
height = h
img = i
return img, width, height
def image_detect(image: TYPE_IMAGE) -> Tuple[TYPE_IMAGE, Tuple[int, ...]]:
gray = image_grayscale(image)
_, thresh = cv2.threshold(gray, 128, 255, cv2.THRESH_BINARY_INV)
# contours
contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
# Assume the largest contour is the item we want to recenter
largest_contour = max(contours, key=cv2.contourArea)
x, y, w, h = cv2.boundingRect(largest_contour)
cropped_image = image[y:y+h, x:x+w]
return cropped_image, (x, y, w, h)
def image_diff(imageA: TYPE_IMAGE, imageB: TYPE_IMAGE, threshold:int=0,
color:TYPE_PIXEL=(255, 0, 0)) -> Tuple[TYPE_IMAGE, TYPE_IMAGE, TYPE_IMAGE, TYPE_IMAGE, float]:
"""imageA, imageB, diff, thresh, score
"""
h1, w1 = imageA.shape[:2]
h2, w2 = imageB.shape[:2]
w1 = max(w1, w2)
h1 = max(h1, h2)
imageA = image_matte(imageA, (0, 0, 0, 0), w1, h1)
imageA = image_convert(imageA, 3)
imageB = image_matte(imageB, (0, 0, 0, 0), w1, h1)
imageB = image_convert(imageB, 3)
grayA = image_grayscale(imageA)
grayB = image_grayscale(imageB)
(score, diff) = ssim(grayA, grayB, full=True, channel_axis=2)
diff = (diff * 255).astype("uint8")
diff_box = cv2.merge([diff, diff, diff])
_, thresh = cv2.threshold(diff, threshold, 255, cv2.THRESH_BINARY_INV | cv2.THRESH_OTSU)
contours = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
contours = contours[0] if len(contours) == 2 else contours[1]
high_a = imageA.copy()
high_a = image_convert(high_a, 3)
high_b = imageB.copy()
high_b = image_convert(high_b, 3)
for c in contours:
area = cv2.contourArea(c)
if area > 40:
x,y,w,h = cv2.boundingRect(c)
cv2.rectangle(imageA, (x, y), (x + w, y + h), (36,255,12), 2)
cv2.rectangle(imageB, (x, y), (x + w, y + h), (36,255,12), 2)
cv2.rectangle(diff_box, (x, y), (x + w, y + h), (36,255,12), 2)
cv2.drawContours(high_a, [c], 0, color[::-1], -1)
cv2.drawContours(high_b, [c], 0, color[::-1], -1)
cv2.drawContours(diff_box, [c], 0, color[::-1], -1)
imageA = cv2.addWeighted(imageA, 0.0, high_a, 1, 0)
imageB = cv2.addWeighted(imageB, 0.0, high_b, 1, 0)
return imageA, imageB, diff, thresh, score
def image_disparity(imageA: np.ndarray) -> np.ndarray:
imageA = imageA.astype(np.float32) / 255.
imageA = cv2.normalize(imageA, None, alpha=0, beta=1, norm_type=cv2.NORM_MINMAX)
disparity_map = np.divide(1.0, imageA, where=imageA != 0)
return np.where(imageA == 0, 1, disparity_map)
def image_gradient_map2(image, gradient_map):
na = np.array(image)
grey = np.mean(na, axis=2).astype(np.uint8)
cmap = np.array(gradient_map.convert('RGB'))
result = np.zeros((*grey.shape, 3), dtype=np.uint8)
grey_reshaped = grey.reshape(-1)
np.take(cmap.reshape(-1, 3), grey_reshaped, axis=0, out=result.reshape(-1, 3))
return result
def image_gradient(width:int, height:int, color_map:dict=None) -> TYPE_IMAGE:
if color_map is None:
color_map = {0: (0,0,0,255)}
@@ -190,216 +104,6 @@ def image_gradient(width:int, height:int, color_map:dict=None) -> TYPE_IMAGE:
draw[x, y] = r, g, b
return pil2cv(image)
# Adapted from WAS Suite -- gradient_map
# https://github.com/WASasquatch/was-node-suite-comfyui
def image_gradient_map(image:TYPE_IMAGE, gradient_map:TYPE_IMAGE, reverse:bool=False) -> TYPE_IMAGE:
if reverse:
gradient_map = gradient_map[:,:,::-1]
grey = image_grayscale(image)
cmap = image_convert(gradient_map, 3)
cmap = cv2.resize(cmap, (256, 256))
cmap = cmap[0,:,:].reshape((256, 1, 3)).astype(np.uint8)
return cv2.applyColorMap(grey, cmap)
def image_grid(data: List[TYPE_IMAGE], width: int, height: int) -> TYPE_IMAGE:
#@TODO: makes poor assumption all images are the same dimensions.
chunks, col, row = grid_make(data)
frame = np.zeros((height * row, width * col, 4), dtype=np.uint8)
i = 0
for y, strip in enumerate(chunks):
for x, item in enumerate(strip):
cc = item.shape[2] if item.ndim == 3 else 1
if cc == 3:
item = channel_add(item)
y1, y2 = y * height, (y+1) * height
x1, x2 = x * width, (x+1) * width
frame[y1:y2, x1:x2, ] = item
i += 1
return frame
def image_histogram(image:TYPE_IMAGE, bins=256) -> TYPE_IMAGE:
bins = max(image.max(), bins) + 1
flatImage = image.flatten()
histogram = np.zeros(bins)
for pixel in flatImage:
histogram[pixel] += 1
return histogram
def image_lerp(imageA: TYPE_IMAGE, imageB:TYPE_IMAGE, mask:TYPE_IMAGE=None,
alpha:float=1.) -> TYPE_IMAGE:
imageA = imageA.astype(np.float32)
imageB = imageB.astype(np.float32)
# establish mask
alpha = np.clip(alpha, 0, 1)
if mask is None:
height, width = imageA.shape[:2]
mask = np.ones((height, width, 1), dtype=np.float32)
else:
# normalize the mask
mask = mask.astype(np.float32)
mask = (mask - mask.min()) / (mask.max() - mask.min()) * alpha
# LERP
imageA = cv2.multiply(1. - mask, imageA)
imageB = cv2.multiply(mask, imageB)
imageA = (cv2.add(imageA, imageB) / 255. - 0.5) * 2.0
imageA = (imageA * 255).astype(np.uint8)
return np.clip(imageA, 0, 255)
def image_levels(image: np.ndarray, black_point:int=0, white_point=255,
mid_point=128, gamma=1.0) -> np.ndarray:
"""
Adjusts the levels of an image including black, white, midpoints, and gamma correction.
Args:
image (numpy.ndarray): Input image tensor in RGB(A) format.
black_point (int): The black point to adjust shadows. Default is 0.
white_point (int): The white point to adjust highlights. Default is 255.
mid_point (int): The mid point for mid-tone adjustment. Default is 128.
gamma (float): Gamma correction value. Default is 1.0.
Returns:
numpy.ndarray: Adjusted image tensor.
"""
image, alpha, cc = image2bgr(image)
# Convert points and gamma to float32 for calculations
black = np.array([black_point] * 3, dtype=np.float32)
white = np.array([white_point] * 3, dtype=np.float32)
mid = np.array([mid_point] * 3, dtype=np.float32)
inGamma = np.array([gamma] * 3, dtype=np.float32)
outBlack = np.array([0, 0, 0], dtype=np.float32)
outWhite = np.array([255, 255, 255], dtype=np.float32)
# Apply levels adjustment
image = np.clip((image - black) / (white - black), 0, 1)
image = (image - mid) / (1.0 - mid)
image = (image ** (1 / inGamma)) * (outWhite - outBlack) + outBlack
image = np.clip(image, 0, 255).astype(np.uint8)
return bgr2image(image, alpha, cc == 1)
def image_merge(imageA: TYPE_IMAGE, imageB: TYPE_IMAGE, axis: int=0,
flip: bool=False) -> TYPE_IMAGE:
if flip:
imageA, imageB = imageB, imageA
axis = 1 if axis == "HORIZONTAL" else 0
return np.concatenate((imageA, imageB), axis=axis)
def image_mirror(image: TYPE_IMAGE, mode:EnumMirrorMode, x:float=0.5,
y:float=0.5) -> TYPE_IMAGE:
cc = image.shape[2] if image.ndim == 3 else 1
height, width = image.shape[:2]
def mirror(img:TYPE_IMAGE, axis:int, reverse:bool=False) -> TYPE_IMAGE:
pivot = x if axis == 1 else y
flip = cv2.flip(img, axis)
pivot = np.clip(pivot, 0, 1)
if reverse:
pivot = 1. - pivot
flip, img = img, flip
scalar = height if axis == 0 else width
slice1 = int(pivot * scalar)
slice1w = scalar - slice1
slice2w = min(scalar - slice1w, slice1w)
if cc >= 3:
output = np.zeros((height, width, cc), dtype=np.uint8)
else:
output = np.zeros((height, width), dtype=np.uint8)
if axis == 0:
output[:slice1, :] = img[:slice1, :]
output[slice1:slice1 + slice2w, :] = flip[slice1w:slice1w + slice2w, :]
else:
output[:, :slice1] = img[:, :slice1]
output[:, slice1:slice1 + slice2w] = flip[:, slice1w:slice1w + slice2w]
return output
if mode in [EnumMirrorMode.X, EnumMirrorMode.FLIP_X, EnumMirrorMode.XY, EnumMirrorMode.FLIP_XY, EnumMirrorMode.X_FLIP_Y, EnumMirrorMode.FLIP_X_FLIP_Y]:
reverse = mode in [EnumMirrorMode.FLIP_X, EnumMirrorMode.FLIP_XY, EnumMirrorMode.FLIP_X_FLIP_Y]
image = mirror(image, 1, reverse)
if mode not in [EnumMirrorMode.NONE, EnumMirrorMode.X, EnumMirrorMode.FLIP_X]:
reverse = mode in [EnumMirrorMode.FLIP_Y, EnumMirrorMode.FLIP_X_FLIP_Y, EnumMirrorMode.X_FLIP_Y]
image = mirror(image, 0, reverse)
return image
def image_mirror_mandela(imageA: np.ndarray, imageB: np.ndarray) -> Tuple[np.ndarray, ...]:
"""Merge 4 flipped copies of input images to make them wrap.
Output is twice bigger in both dimensions."""
top = np.hstack([imageA, -np.flip(imageA, axis=1)])
bottom = np.hstack([np.flip(imageA, axis=0), -np.flip(imageA)])
imageA = np.vstack([top, bottom])
top = np.hstack([imageB, np.flip(imageB, axis=1)])
bottom = np.hstack([-np.flip(imageB, axis=0), -np.flip(imageB)])
imageB = np.vstack([top, bottom])
return imageA, imageB
def image_pixelate(image: TYPE_IMAGE, amount:float=1.)-> TYPE_IMAGE:
h, w = image.shape[:2]
amount = max(0, min(1, amount))
block_size_h = max(1, (h * amount))
block_size_w = max(1, (w * amount))
num_blocks_h = int(np.ceil(h / block_size_h))
num_blocks_w = int(np.ceil(w / block_size_w))
block_size_h = h // num_blocks_h
block_size_w = w // num_blocks_w
pixelated_image = image.copy()
for i in range(num_blocks_h):
for j in range(num_blocks_w):
# Calculate block boundaries
y_start = i * block_size_h
y_end = min((i + 1) * block_size_h, h)
x_start = j * block_size_w
x_end = min((j + 1) * block_size_w, w)
# Average color values within the block
block_average = np.mean(image[y_start:y_end, x_start:x_end], axis=(0, 1))
# Fill the block with the average color
pixelated_image[y_start:y_end, x_start:x_end] = block_average
return pixelated_image.astype(np.uint8)
def image_posterize(image: TYPE_IMAGE, levels:int=256) -> TYPE_IMAGE:
divisor = 256 / max(2, min(256, levels))
return (np.floor(image / divisor) * int(divisor)).astype(np.uint8)
def image_quantize(image:TYPE_IMAGE, levels:int=256, iterations:int=10,
epsilon:float=0.2) -> TYPE_IMAGE:
levels = int(max(2, min(256, levels)))
pixels = np.float32(image)
criteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, iterations, epsilon)
_, labels, centers = cv2.kmeans(pixels, levels, None, criteria, 5, cv2.KMEANS_RANDOM_CENTERS)
centers = np.uint8(centers)
return centers[labels.flatten()].reshape(image.shape)
def image_sharpen(image:TYPE_IMAGE, kernel_size=None, sigma:float=1.0,
amount:float=1.0, threshold:float=0) -> TYPE_IMAGE:
"""Return a sharpened version of the image, using an unsharp mask."""
kernel_size = (kernel_size, kernel_size) if kernel_size else (5, 5)
blurred = cv2.GaussianBlur(image, kernel_size, sigma)
sharpened = float(amount + 1) * image - float(amount) * blurred
sharpened = np.maximum(sharpened, np.zeros(sharpened.shape))
sharpened = np.minimum(sharpened, 255 * np.ones(sharpened.shape))
sharpened = sharpened.round().astype(np.uint8)
if threshold > 0:
low_contrast_mask = np.absolute(image - blurred) < threshold
np.copyto(sharpened, image, where=low_contrast_mask)
return sharpened
def image_split(image: TYPE_IMAGE) -> Tuple[TYPE_IMAGE, ...]:
h, w = image.shape[:2]
@@ -416,45 +120,6 @@ def image_split(image: TYPE_IMAGE) -> Tuple[TYPE_IMAGE, ...]:
r, g, b, a = cv2.split(image)
return r, g, b, a
def image_stack(image_list: List[TYPE_IMAGE],
axis:EnumOrientation=EnumOrientation.HORIZONTAL,
stride:int=0, matte:TYPE_PIXEL=(0,0,0,255)) -> TYPE_IMAGE:
_, width, height = image_by_size(image_list)
images = [image_matte(image_convert(i, 4), matte, width, height) for i in image_list]
count = len(images)
matte = pixel_convert(matte, 4)
match axis:
case EnumOrientation.GRID:
if stride < 1:
stride = np.ceil(np.sqrt(count))
stride = int(stride)
stride = min(stride, count)
stride = max(stride, 1)
rows = []
for i in range(0, count, stride):
row = images[i:i + stride]
row_stacked = np.hstack(row)
rows.append(row_stacked)
height, width = images[0].shape[:2]
overhang = count % stride
if overhang != 0:
overhang = stride - overhang
size = (height, overhang * width, 4)
filler = np.full(size, matte, dtype=np.uint8)
rows[-1] = np.hstack([rows[-1], filler])
image = np.vstack(rows)
case EnumOrientation.HORIZONTAL:
image = np.hstack(images)
case EnumOrientation.VERTICAL:
image = np.vstack(images)
return image
def image_stereogram(image: TYPE_IMAGE, depth: TYPE_IMAGE, divisions:int=8,
mix:float=0.33, gamma:float=0.33, shift:float=1.) -> TYPE_IMAGE:
height, width = depth.shape[:2]
@@ -480,31 +145,6 @@ def image_stereogram(image: TYPE_IMAGE, depth: TYPE_IMAGE, divisions:int=8,
out[y, x] = out[y, pos]
return out
def image_stereo_shift(image: TYPE_IMAGE, depth: TYPE_IMAGE, shift:float=10) -> TYPE_IMAGE:
# Ensure base image has alpha
image = image_convert(image, 4)
depth = image_convert(depth, 1)
deltas = np.array((depth / 255.0) * float(shift), dtype=int)
shifted_data = np.zeros(image.shape, dtype=np.uint8)
_, width = image.shape[:2]
for y, row in enumerate(deltas):
for x, dx in enumerate(row):
x2 = x + dx
if (x2 >= width) or (x2 < 0):
continue
shifted_data[y][x2] = image[y][x]
shifted_image = cv2pil(shifted_data)
alphas_image = Image.fromarray(
ndimage.binary_fill_holes(
ImageChops.invert(
shifted_image.getchannel("A")
)
)
).convert("1")
shifted_image.putalpha(ImageChops.invert(alphas_image))
return pil2cv(shifted_image)
def image_threshold(image:TYPE_IMAGE, threshold:float=0.5,
mode:EnumThreshold=EnumThreshold.BINARY,
adapt:EnumThresholdAdapt=EnumThresholdAdapt.ADAPT_NONE,
@@ -545,267 +185,3 @@ def morph_emboss(image: TYPE_IMAGE, amount: float=1., kernel: int=2) -> TYPE_IMA
[kernel-2, kernel-1, 2]
]) * amount
return cv2.filter2D(src=image, ddepth=-1, kernel=kernel)
# KERNELS
def MEDIAN3x3(image: TYPE_IMAGE) -> TYPE_IMAGE:
height, width = image.shape[:2]
out = np.zeros([height, width])
for i in range(1, height-1):
for j in range(1, width-1):
temp = [
image[i-1, j-1],
image[i-1, j],
image[i-1, j + 1],
image[i, j-1],
image[i, j],
image[i, j + 1],
image[i + 1, j-1],
image[i + 1, j],
image[i + 1, j + 1]
]
temp = sorted(temp)
out[i, j]= temp[4]
return out
def kernel(stride: int) -> TYPE_IMAGE:
"""
Generate a kernel matrix with a specific stride.
The kernel matrix has a size of (stride, stride) and is filled with values
such that if i < j, the element is set to -1; if i > j, the element is set to 1.
Parameters:
- stride (int): The size of the square kernel matrix.
Returns:
- TYPE_IMAGE: The generated kernel matrix.
Example:
>>> KERNEL(3)
array([[ 0, 1, 1],
[-1, 0, 1],
[-1, -1, 0]], dtype=int8)
"""
# Create an initial matrix of zeros
kernel = np.zeros((stride, stride), dtype=np.int8)
# Create a mask for elements where i < j and set them to -1
mask_lower = np.tril(np.ones((stride, stride), dtype=bool), k=-1)
kernel[mask_lower] = -1
# Create a mask for elements where i > j and set them to 1
mask_upper = np.triu(np.ones((stride, stride), dtype=bool), k=1)
kernel[mask_upper] = 1
return kernel
# =============================================================================
def coord_cart2polar(x: float, y: float) -> TYPE_fCOORD2D:
r = np.sqrt(x**2 + y**2)
theta = np.arctan2(y, x)
return r, theta
def coord_polar2cart(r: float, theta: float) -> TYPE_fCOORD2D:
x = r * np.cos(theta)
y = r * np.sin(theta)
return x, y
def coord_default(width:int, height:int, origin:TYPE_fCOORD2D=None) -> TYPE_fCOORD2D:
"""Creates x & y coords for the indicies in a numpy array "data".
"origin" defaults to the center of the image. Specify origin=(0,0)
to set the origin to the lower left corner of the image."""
if origin is None:
origin_x, origin_y = width // 2, height // 2
else:
origin_x, origin_y = origin
x, y = np.meshgrid(np.arange(width), np.arange(height))
x -= origin_x
y -= origin_y
return x, y
def coord_fisheye(width: int, height: int, distortion: float) -> Tuple[TYPE_IMAGE, TYPE_IMAGE]:
map_x, map_y = np.meshgrid(np.linspace(0., 1., width), np.linspace(0., 1., height))
# normalized
xnd, ynd = (2 * map_x - 1), (2 * map_y - 1)
rd = np.sqrt(xnd**2 + ynd**2)
# fish-eye distortion
condition = (dist := 1 - distortion * (rd**2)) == 0
xdu, ydu = np.where(condition, xnd, xnd / dist), np.where(condition, ynd, ynd / dist)
xu, yu = ((xdu + 1) * width) / 2, ((ydu + 1) * height) / 2
return xu.astype(np.float32), yu.astype(np.float32)
def coord_perspective(width: int, height: int, pts: List[TYPE_fCOORD2D]) -> TYPE_IMAGE:
object_pts = np.float32([[0, 0], [width, 0], [width, height], [0, height]])
pts = np.float32(pts)
pts = np.column_stack([pts[:, 0], pts[:, 1]])
return cv2.getPerspectiveTransform(object_pts, pts)
def coord_sphere(width: int, height: int, radius: float) -> Tuple[TYPE_IMAGE, TYPE_IMAGE]:
theta, phi = np.meshgrid(np.linspace(0, TAU, width), np.linspace(0, np.pi, height))
x = radius * np.sin(phi) * np.cos(theta)
y = radius * np.sin(phi) * np.sin(theta)
# z = radius * np.cos(phi)
x_image = (x + 1) * (width - 1) / 2
y_image = (y + 1) * (height - 1) / 2
return x_image.astype(np.float32), y_image.astype(np.float32)
def remap_fisheye(image: TYPE_IMAGE, distort: float) -> TYPE_IMAGE:
cc = image.shape[2] if image.ndim == 3 else 1
height, width = image.shape[:2]
if cc == 1:
image = cv2.cvtColor(image, cv2.COLOR_GRAY2BGR)
map_x, map_y = coord_fisheye(width, height, distort)
image = cv2.remap(image, map_x, map_y, interpolation=cv2.INTER_LINEAR, borderMode=cv2.BORDER_CONSTANT)
#if cc == 1:
# image = image[..., 0]
return image
def remap_perspective(image: TYPE_IMAGE, pts: list) -> TYPE_IMAGE:
cc = image.shape[2] if image.ndim == 3 else 1
height, width = image.shape[:2]
if cc == 1:
image = cv2.cvtColor(image, cv2.COLOR_GRAY2BGR)
pts = coord_perspective(width, height, pts)
image = cv2.warpPerspective(image, pts, (width, height))
#if cc == 1:
# image = image[..., 0]
return image
def remap_polar(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Re-projects a 3D numpy array ("data") into a polar coordinate system.
"origin" is a tuple of (x0, y0) and defaults to the center of the image."""
h, w = image.shape[:2]
radius = max(w, h)
return cv2.linearPolar(image, (h // 2, w // 2), radius // 2, cv2.WARP_INVERSE_MAP)
def remap_sphere(image: TYPE_IMAGE, radius: float) -> TYPE_IMAGE:
height, width = image.shape[:2]
map_x, map_y = coord_sphere(width, height, radius)
return cv2.remap(image, map_x, map_y, interpolation=cv2.INTER_LINEAR, borderMode=cv2.BORDER_CONSTANT)
def depth_from_gradient(grad_x, grad_y):
"""Optimized Frankot-Chellappa depth-from-gradient algorithm."""
rows, cols = grad_x.shape
rows_scale = np.fft.fftfreq(rows)
cols_scale = np.fft.fftfreq(cols)
u_grid, v_grid = np.meshgrid(cols_scale, rows_scale)
grad_x_F = np.fft.fft2(grad_x)
grad_y_F = np.fft.fft2(grad_y)
denominator = u_grid**2 + v_grid**2
denominator[0, 0] = 1.0
Z_F = (-1j * u_grid * grad_x_F - 1j * v_grid * grad_y_F) / denominator
Z_F[0, 0] = 0.0
Z = np.fft.ifft2(Z_F).real
Z -= np.min(Z)
Z /= np.max(Z)
return Z
def height_from_normal(image: TYPE_IMAGE, tile:bool=True) -> TYPE_IMAGE:
"""Computes a height map from the given normal map."""
image = np.transpose(image, (2, 0, 1))
flip_img = np.flip(image, axis=1)
grad_x, grad_y = (flip_img[0] - 0.5) * 2, (flip_img[1] - 0.5) * 2
grad_x = np.flip(grad_x, axis=0)
grad_y = np.flip(grad_y, axis=0)
if not tile:
grad_x, grad_y = image_mirror_mandela(grad_x, grad_y)
pred_img = depth_from_gradient(-grad_x, grad_y)
# re-crop
if not tile:
height, width = image.shape[1], image.shape[2]
pred_img = pred_img[:height, :width]
image = np.stack([pred_img, pred_img, pred_img])
image = np.transpose(image, (1, 2, 0))
return image
def curvature_from_normal(image: TYPE_IMAGE, blur_radius:int=2)-> TYPE_IMAGE:
"""Computes a curvature map from the given normal map."""
image = np.transpose(image, (2, 0, 1))
blur_factor = 1 / 2 ** min(8, max(2, blur_radius))
diff_kernel = np.array([-1, 0, 1])
def conv_1d(array, kernel) -> np.ndarray[Any, np.dtype[Any]]:
"""Performs row-wise 1D convolutions with repeat padding."""
k_l = len(kernel)
extended = np.pad(array, k_l // 2, mode="wrap")
return np.array([np.convolve(row, kernel, mode="valid") for row in extended[k_l//2:-k_l//2+1]])
h_conv = conv_1d(image[0], diff_kernel)
v_conv = conv_1d(-image[1].T, diff_kernel).T
edges_conv = h_conv + v_conv
# Calculate blur radius in pixels
blur_radius_px = int(np.mean(image.shape[1:3]) * blur_factor)
if blur_radius_px < 2:
# If blur radius is too small, just normalize the edge convolution
image = (edges_conv - np.min(edges_conv)) / (np.ptp(edges_conv) + 1e-10)
else:
blur_radius_px += blur_radius_px % 2 == 0
# Compute Gaussian kernel
sigma = max(1, blur_radius_px // 8)
x = np.linspace(-(blur_radius_px - 1) / 2, (blur_radius_px - 1) / 2, blur_radius_px)
g_kernel = np.exp(-0.5 * np.square(x) / np.square(sigma))
g_kernel /= np.sum(g_kernel)
# Apply Gaussian blur
h_blur = conv_1d(edges_conv, g_kernel)
v_blur = conv_1d(h_blur.T, g_kernel).T
image = (v_blur - np.min(v_blur)) / (np.ptp(v_blur) + 1e-10)
image = (image - image.min()) / (image.max() - image.min()) * 255
return image.astype(np.uint8)
def roughness_from_normal(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Roughness from a normal map."""
up_vector = np.array([0, 0, 1])
image = 1 - np.dot(image, up_vector)
image = (image - image.min()) / (image.max() - image.min())
image = (255 * image).astype(np.uint8)
return image_grayscale(image)
def roughness_from_albedo(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""Roughness from an albedo map."""
kernel_size = 3
image = cv2.Laplacian(image, cv2.CV_64F, ksize=kernel_size)
image = (image - image.min()) / (image.max() - image.min())
image = (255 * image).astype(np.uint8)
return image_grayscale(image)
def roughness_from_albedo_normal(albedo: TYPE_IMAGE, normal: TYPE_IMAGE,
blur:int=2, blend:float=0.5, iterations:int=3) -> TYPE_IMAGE:
normal = roughness_from_normal(normal)
normal = image_normalize(normal)
albedo = roughness_from_albedo(albedo)
albedo = image_normalize(albedo)
rough = image_lerp(normal, albedo, alpha=blend)
rough = image_normalize(rough)
image = image_lerp(normal, rough, alpha=blend)
iterations = min(16, max(2, iterations))
blur += (blur % 2 == 0)
step = 1 / 2 ** iterations
for i in range(iterations):
image = cv2.add(normal * step, image * step)
image = cv2.GaussianBlur(image, (blur + i * 2, blur + i * 2), 3 * i)
inverted = 255 - image_normalize(image)
inverted = cv2.subtract(inverted, albedo) * 0.5
inverted = cv2.GaussianBlur(inverted, (blur, blur), blur)
inverted = image_normalize(inverted)
image = cv2.add(image * 0.5, inverted * 0.5)
for i in range(iterations):
image = cv2.GaussianBlur(image, (blur, blur), blur)
image = cv2.add(image * 0.5, inverted * 0.5)
for i in range(iterations):
image = cv2.GaussianBlur(image, (blur, blur), blur)
image = image_normalize(image)
return image
+440
View File
@@ -0,0 +1,440 @@
import urllib
from typing import List, Tuple
import cv2
import numpy as np
import requests
from scipy import ndimage
from skimage.metrics import structural_similarity as ssim
from PIL import Image, ImageChops, ImageOps
from loguru import logger
from Jovimetrix.sup.image import TYPE_IMAGE, TYPE_PIXEL, cv2pil, image_convert, \
image_grayscale, image_matte, pil2cv
from Jovimetrix.sup.image.channel import channel_add
from Jovimetrix.sup.util import grid_make
def image_crop_head(image: TYPE_IMAGE) -> TYPE_IMAGE:
"""
Given a file path or np.ndarray image with a face,
returns cropped np.ndarray around the largest detected
face.
Parameters
----------
- `path_or_array` : {`str`, `np.ndarray`}
* The filepath or numpy array of the image.
Returns
-------
- `image` : {`np.ndarray`, `None`}
* A cropped numpy array if face detected, else None.
"""
MIN_FACE = 8
gray = image_grayscale(image)
h, w = image.shape[:2]
minface = int(np.sqrt(h**2 + w**2) / MIN_FACE)
'''
# Create the haar cascade
face_cascade = cv2.CascadeClassifier(self.casc_path)
# ====== Detect faces in the image ======
faces = face_cascade.detectMultiScale(
gray,
scaleFactor=1.1,
minNeighbors=5,
minSize=(minface, minface),
flags=cv2.CASCADE_FIND_BIGGEST_OBJECT | cv2.CASCADE_DO_ROUGH_SEARCH,
)
# Handle no faces
if len(faces) == 0:
return None
# Make padding from biggest face found
x, y, w, h = faces[-1]
pos = self._crop_positions(
img_height,
img_width,
x,
y,
w,
h,
)
# ====== Actual cropping ======
image = image[pos[0] : pos[1], pos[2] : pos[3]]
# Resize
if self.resize:
with Image.fromarray(image) as img:
image = np.array(img.resize((self.width, self.height)))
# Underexposition fix
if self.gamma:
image = check_underexposed(image, gray)
return bgr_to_rbg(image)
def _determine_safe_zoom(self, imgh, imgw, x, y, w, h):
"""
Determines the safest zoom level with which to add margins
around the detected face. Tries to honor `self.face_percent`
when possible.
Parameters:
-----------
imgh: int
Height (px) of the image to be cropped
imgw: int
Width (px) of the image to be cropped
x: int
Leftmost coordinates of the detected face
y: int
Bottom-most coordinates of the detected face
w: int
Width of the detected face
h: int
Height of the detected face
Diagram:
--------
i / j := zoom / 100
+
h1 | h2
+---------|---------+
| MAR|GIN |
| (x+w, y+h)|
| +-----|-----+ |
| | FA|CE | |
| | | | |
| ├──i──┤ | |
| | cen|ter | |
| | | | |
| +-----|-----+ |
| (x, y)| |
| | |
+---------|---------+
├────j────┤
+
"""
# Find out what zoom factor to use given self.aspect_ratio
corners = itertools.product((x, x + w), (y, y + h))
center = np.array([x + int(w / 2), y + int(h / 2)])
i = np.array(
[(0, 0), (0, imgh), (imgw, imgh), (imgw, 0), (0, 0)]
) # image_corners
image_sides = [(i[n], i[n + 1]) for n in range(4)]
corner_ratios = [self.face_percent] # Hopefully we use this one
for c in corners:
corner_vector = np.array([center, c])
a = distance(*corner_vector)
intersects = list(intersect(corner_vector, side) for side in image_sides)
for pt in intersects:
if (pt >= 0).all() and (pt <= i[2]).all(): # if intersect within image
dist_to_pt = distance(center, pt)
corner_ratios.append(100 * a / dist_to_pt)
return max(corner_ratios)
def _crop_positions(
self,
imgh,
imgw,
x,
y,
w,
h,
):
"""
Retuns the coordinates of the crop position centered
around the detected face with extra margins. Tries to
honor `self.face_percent` if possible, else uses the
largest margins that comply with required aspect ratio
given by `self.height` and `self.width`.
Parameters:
-----------
imgh: int
Height (px) of the image to be cropped
imgw: int
Width (px) of the image to be cropped
x: int
Leftmost coordinates of the detected face
y: int
Bottom-most coordinates of the detected face
w: int
Width of the detected face
h: int
Height of the detected face
"""
zoom = self._determine_safe_zoom(imgh, imgw, x, y, w, h)
# Adjust output height based on percent
if self.height >= self.width:
height_crop = h * 100.0 / zoom
width_crop = self.aspect_ratio * float(height_crop)
else:
width_crop = w * 100.0 / zoom
height_crop = float(width_crop) / self.aspect_ratio
# Calculate padding by centering face
xpad = (width_crop - w) / 2
ypad = (height_crop - h) / 2
# Calc. positions of crop
h1 = x - xpad
h2 = x + w + xpad
v1 = y - ypad
v2 = y + h + ypad
return [int(v1), int(v2), int(h1), int(h2)]
'''
def image_detect(image: TYPE_IMAGE) -> Tuple[TYPE_IMAGE, Tuple[int, ...]]:
gray = image_grayscale(image)
_, thresh = cv2.threshold(gray, 128, 255, cv2.THRESH_BINARY_INV)
# contours
contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
# Assume the largest contour is the item we want to recenter
largest_contour = max(contours, key=cv2.contourArea)
x, y, w, h = cv2.boundingRect(largest_contour)
cropped_image = image[y:y+h, x:x+w]
return cropped_image, (x, y, w, h)
def image_diff(imageA: TYPE_IMAGE, imageB: TYPE_IMAGE, threshold:int=0,
color:TYPE_PIXEL=(255, 0, 0)) -> Tuple[TYPE_IMAGE, TYPE_IMAGE, TYPE_IMAGE, TYPE_IMAGE, float]:
"""imageA, imageB, diff, thresh, score
"""
h1, w1 = imageA.shape[:2]
h2, w2 = imageB.shape[:2]
w1 = max(w1, w2)
h1 = max(h1, h2)
imageA = image_matte(imageA, (0, 0, 0, 0), w1, h1)
imageA = image_convert(imageA, 3)
imageB = image_matte(imageB, (0, 0, 0, 0), w1, h1)
imageB = image_convert(imageB, 3)
grayA = image_grayscale(imageA)
grayB = image_grayscale(imageB)
(score, diff) = ssim(grayA, grayB, full=True, channel_axis=2)
diff = (diff * 255).astype("uint8")
diff_box = cv2.merge([diff, diff, diff])
_, thresh = cv2.threshold(diff, threshold, 255, cv2.THRESH_BINARY_INV | cv2.THRESH_OTSU)
contours = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
contours = contours[0] if len(contours) == 2 else contours[1]
high_a = imageA.copy()
high_a = image_convert(high_a, 3)
high_b = imageB.copy()
high_b = image_convert(high_b, 3)
for c in contours:
area = cv2.contourArea(c)
if area > 40:
x,y,w,h = cv2.boundingRect(c)
cv2.rectangle(imageA, (x, y), (x + w, y + h), (36,255,12), 2)
cv2.rectangle(imageB, (x, y), (x + w, y + h), (36,255,12), 2)
cv2.rectangle(diff_box, (x, y), (x + w, y + h), (36,255,12), 2)
cv2.drawContours(high_a, [c], 0, color[::-1], -1)
cv2.drawContours(high_b, [c], 0, color[::-1], -1)
cv2.drawContours(diff_box, [c], 0, color[::-1], -1)
imageA = cv2.addWeighted(imageA, 0.0, high_a, 1, 0)
imageB = cv2.addWeighted(imageB, 0.0, high_b, 1, 0)
return imageA, imageB, diff, thresh, score
def image_disparity(imageA: np.ndarray) -> np.ndarray:
imageA = imageA.astype(np.float32) / 255.
imageA = cv2.normalize(imageA, None, alpha=0, beta=1, norm_type=cv2.NORM_MINMAX)
disparity_map = np.divide(1.0, imageA, where=imageA != 0)
return np.where(imageA == 0, 1, disparity_map)
def image_histogram_statistics(histogram:np.ndarray, L=256)-> TYPE_IMAGE:
sumPixels = np.sum(histogram)
normalizedHistogram = histogram/sumPixels
mean = 0
for i in range(L):
mean += i * normalizedHistogram[i]
variance = 0
for i in range(L):
variance += (i-mean)**2 * normalizedHistogram[i]
std = np.sqrt(variance)
return mean, variance, std
def image_gradient_map2(image, gradient_map):
na = np.array(image)
grey = np.mean(na, axis=2).astype(np.uint8)
cmap = np.array(gradient_map.convert('RGB'))
result = np.zeros((*grey.shape, 3), dtype=np.uint8)
grey_reshaped = grey.reshape(-1)
np.take(cmap.reshape(-1, 3), grey_reshaped, axis=0, out=result.reshape(-1, 3))
return result
def image_grid(data: List[TYPE_IMAGE], width: int, height: int) -> TYPE_IMAGE:
#@TODO: makes poor assumption all images are the same dimensions.
chunks, col, row = grid_make(data)
frame = np.zeros((height * row, width * col, 4), dtype=np.uint8)
i = 0
for y, strip in enumerate(chunks):
for x, item in enumerate(strip):
cc = item.shape[2] if item.ndim == 3 else 1
if cc == 3:
item = channel_add(item)
y1, y2 = y * height, (y+1) * height
x1, x2 = x * width, (x+1) * width
frame[y1:y2, x1:x2, ] = item
i += 1
return frame
def image_merge(imageA: TYPE_IMAGE, imageB: TYPE_IMAGE, axis: int=0,
flip: bool=False) -> TYPE_IMAGE:
if flip:
imageA, imageB = imageB, imageA
axis = 1 if axis == "HORIZONTAL" else 0
return np.concatenate((imageA, imageB), axis=axis)
def image_recenter(image: TYPE_IMAGE) -> TYPE_IMAGE:
cropped_image = image_detect(image)[0]
new_image = np.zeros(image.shape, dtype=np.uint8)
paste_x = (new_image.shape[1] - cropped_image.shape[1]) // 2
paste_y = (new_image.shape[0] - cropped_image.shape[0]) // 2
new_image[paste_y:paste_y+cropped_image.shape[0], paste_x:paste_x+cropped_image.shape[1]] = cropped_image
return new_image
def image_stereo_shift(image: TYPE_IMAGE, depth: TYPE_IMAGE, shift:float=10) -> TYPE_IMAGE:
# Ensure base image has alpha
image = image_convert(image, 4)
depth = image_convert(depth, 1)
deltas = np.array((depth / 255.0) * float(shift), dtype=int)
shifted_data = np.zeros(image.shape, dtype=np.uint8)
_, width = image.shape[:2]
for y, row in enumerate(deltas):
for x, dx in enumerate(row):
x2 = x + dx
if (x2 >= width) or (x2 < 0):
continue
shifted_data[y][x2] = image[y][x]
shifted_image = cv2pil(shifted_data)
alphas_image = Image.fromarray(
ndimage.binary_fill_holes(
ImageChops.invert(
shifted_image.getchannel("A")
)
)
).convert("1")
shifted_image.putalpha(ImageChops.invert(alphas_image))
return pil2cv(shifted_image)
# KERNELS
def MEDIAN3x3(image: TYPE_IMAGE) -> TYPE_IMAGE:
height, width = image.shape[:2]
out = np.zeros([height, width])
for i in range(1, height-1):
for j in range(1, width-1):
temp = [
image[i-1, j-1],
image[i-1, j],
image[i-1, j + 1],
image[i, j-1],
image[i, j],
image[i, j + 1],
image[i + 1, j-1],
image[i + 1, j],
image[i + 1, j + 1]
]
temp = sorted(temp)
out[i, j]= temp[4]
return out
def kernel(stride: int) -> TYPE_IMAGE:
"""
Generate a kernel matrix with a specific stride.
The kernel matrix has a size of (stride, stride) and is filled with values
such that if i < j, the element is set to -1; if i > j, the element is set to 1.
Parameters:
- stride (int): The size of the square kernel matrix.
Returns:
- TYPE_IMAGE: The generated kernel matrix.
Example:
>>> KERNEL(3)
array([[ 0, 1, 1],
[-1, 0, 1],
[-1, -1, 0]], dtype=int8)
"""
# Create an initial matrix of zeros
kernel = np.zeros((stride, stride), dtype=np.int8)
# Create a mask for elements where i < j and set them to -1
mask_lower = np.tril(np.ones((stride, stride), dtype=bool), k=-1)
kernel[mask_lower] = -1
# Create a mask for elements where i > j and set them to 1
mask_upper = np.triu(np.ones((stride, stride), dtype=bool), k=1)
kernel[mask_upper] = 1
return kernel
#
#
#
def image_load_exr(url: str) -> Tuple[TYPE_IMAGE, TYPE_IMAGE]:
"""
exr_file = OpenEXR.InputFile(url)
exr_header = exr_file.header()
r,g,b = exr_file.channels("RGB", pixel_type=Imath.PixelType(Imath.PixelType.FLOAT) )
dw = exr_header[ "dataWindow" ]
w = dw.max.x - dw.min.x + 1
h = dw.max.y - dw.min.y + 1
image = np.ones( (h, w, 4), dtype = np.float32 )
image[:, :, 0] = np.core.multiarray.frombuffer( r, dtype = np.float32 ).reshape(h, w)
image[:, :, 1] = np.core.multiarray.frombuffer( g, dtype = np.float32 ).reshape(h, w)
image[:, :, 2] = np.core.multiarray.frombuffer( b, dtype = np.float32 ).reshape(h, w)
return create_optix_image_2D( w, h, image.flatten() )
"""
pass
def image_load_from_url(url: str, stream:bool=True) -> TYPE_IMAGE:
"""Creates a CV2 BGR image from a url."""
try:
image = urllib.request.urlopen(url)
image = np.asarray(bytearray(image.read()), dtype=np.uint8)
return cv2.imdecode(image, cv2.IMREAD_COLOR)
except:
try:
image = Image.open(requests.get(url, stream=stream).raw)
return pil2cv(image)
except Exception as e:
logger.error(str(e))
def image_save_gif(fpath:str, images: List[Image.Image], fps: int=0,
loop:int=0, optimize:bool=False) -> None:
fps = min(50, max(1, fps))
images[0].save(
fpath,
append_images=images[1:],
duration=3, # int(100.0 / fps),
loop=loop,
optimize=optimize,
save_all=True
)
def image_load_data(data: str) -> TYPE_IMAGE:
img = ImageOps.exif_transpose(data)
return pil2cv(img)
+1 -1
View File
@@ -9,7 +9,7 @@ import os
import re
import sys
from pathlib import Path
from enum import Enum, EnumType
from enum import Enum, EnumMeta as EnumType
from typing import Any, Dict, Tuple
import cv2
+2 -2
View File
@@ -316,8 +316,8 @@ def parse_param(data:dict, key:str, typ:EnumConvertType, default: Any,
elif isinstance(val, (torch.Tensor,)):
if val.ndim > 3:
val = [t for t in val]
elif val.ndim == 3:
val = [v.unsqueeze(-1) for v in val]
else:
val = [val]
elif isinstance(val, (list, tuple, set)):
if len(val) == 0:
val = [None]