Files
Amorano-Jovi_Capture/core/node_monitor.py
T
Alexander G. Morano 335ae69a71 added flip and reverse
fix camera scan for first time with no auto-scan
2025-02-25 04:24:50 -05:00

146 lines
5.6 KiB
Python

"""Capture Monitors"""
import time
from typing import Dict
import cv2
import mss
import torch
import numpy as np
from PIL import ImageGrab
from comfy.utils import ProgressBar
from cozy_comfyui import \
RGBAMaskType, EnumConvertType, \
logger, \
deep_merge, parse_param, zip_longest_fill
from cozy_comfyui.image import ImageType
from cozy_comfyui.image.convert import cv_to_tensor_full
from . import StreamNodeHeader
# ==============================================================================
# === CONSTANT ===
# ==============================================================================
JOV_DOCKERENV = False
try:
with open('/proc/1/cgroup', 'rt') as f:
content = f.read()
JOV_DOCKERENV = any(x in content for x in ['docker', 'kubepods', 'containerd'])
except FileNotFoundError:
pass
if JOV_DOCKERENV:
logger.info("RUNNING IN A DOCKER")
# ==============================================================================
# === SUPPORT ===
# ==============================================================================
def monitor_capture_all(width:int=None, height:int=None) -> ImageType:
if JOV_DOCKERENV:
return None
img = ImageGrab.grab(all_screens=True)
img = np.array(img, dtype='uint8')
img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)
if height is not None and width is not None:
return cv2.resize(img, (width, height))
return img
# ==============================================================================
# === NODE ===
# ==============================================================================
class MonitorStreamReader(StreamNodeHeader):
NAME = "MONITOR"
DESCRIPTION = """
Capture frames from a desktop monitor. Supports batch processing, allowing multiple frames to be captured simultaneously. The node provides options for configuring the source, resolution, frame rate, zoom, orientation, and interpolation method. Additionally, it supports capturing frames from multiple monitors or windows simultaneously.
"""
MONITOR = None
@classmethod
def INPUT_TYPES(cls) -> Dict[str, str]:
if not JOV_DOCKERENV:
cls.MONITOR = []
with mss.mss() as screen:
for i, m in enumerate(screen.monitors):
cls.MONITOR.append(f"{i}-{m['width']}x{m['height']}")
if cls.MONITOR is None or len(cls.MONITOR) == 0:
cls.MONITOR = ["NONE"]
d = super().INPUT_TYPES()
return deep_merge({
"optional": {
"MONITOR": (cls.MONITOR, {"default": cls.MONITOR[0], "choice": "list of system monitor devices", "tooltip": "list of system monitor devices"}),
"XY": ("VEC2INT", {"default": (0, 0), "mij": 0, "label": ["TOP", "LEFT"], "tooltip": "Top, Left position"}),
"WH": ("VEC2INT", {"default": (0, 0), "mij": 0, "label": ["WIDTH", "HEIGHT"], "tooltip": "Width and Height"})
}
}, d)
def run(self, **kw) -> RGBAMaskType:
if JOV_DOCKERENV:
img = cv_to_tensor_full(self.empty)
return [torch.stack(i) for i in zip(*img)]
# only allow monitor to capture single one per "batch"
images = []
batch_size = parse_param(kw, "BATCH", EnumConvertType.INT, 1, 1)[0]
# allow these to "flex" length so as to animate
monitor = parse_param(kw, "MONITOR", EnumConvertType.STRING, "NONE")
fps = parse_param(kw, "FPS", EnumConvertType.INT, 30)
xy = parse_param(kw, "XY", EnumConvertType.VEC2INT, [(0,0)], 0)
wh = parse_param(kw, "WH", EnumConvertType.VEC2INT, [(0,0)], 0)
flip = parse_param(kw, "FLIP", EnumConvertType.BOOLEAN, False)
reverse = parse_param(kw, "REVERSE", EnumConvertType.BOOLEAN, False)
pbar = ProgressBar(batch_size)
size = [batch_size] * batch_size
params = list(zip_longest_fill(monitor, fps, xy, wh, flip, reverse, size))
with mss.mss() as screen:
for idx, (monitor, fps, xy, wh, flip, reverse, size) in enumerate(params):
try:
monitor = int(monitor.split('-')[0].strip())
except Exception:
logger.warning(f"bad monitor {monitor}")
img = self.empty
else:
capture = screen.monitors[monitor]
width = capture['width']
height = capture['height']
# clip the position to be "in-bounds"
left = min(width-1, xy[0])
top = min(height-1, xy[1])
width = width if wh[0] == 0 else int(np.clip(wh[0], 1, width))
height = height if wh[1] == 0 else int(np.clip(wh[1], 1, height))
width = min(width, capture['width']-left)
height = min(height, capture['height']-top)
region = {
'top': top + capture['top'],
'left': left + capture['left'],
'width': width,
'height': height
}
img = screen.grab(region)
img = cv2.cvtColor(np.array(img, dtype=np.uint8), cv2.COLOR_RGB2BGR)
if flip:
img = cv2.flip(img, 0)
if reverse:
img = cv2.flip(img, 1)
images.append(cv_to_tensor_full(img))
pbar.update_absolute(idx)
if batch_size > 1:
rate = 1. / fps
time.sleep(rate)
return [torch.stack(i) for i in zip(*images)]