17 Commits
Author SHA1 Message Date
PSchroedl 8e6213cfdd Merge pull request #8 from eliteprox/revert-patch1
Revert "update pyproject.toml"
2025-01-20 21:29:03 -08:00
Elite Encoder 6a68c7b11d Revert "update pyproject.toml"
This reverts commit 5465385d48.
2025-01-21 00:16:36 -05:00
John Mull 5465385d48 update pyproject.toml 2025-01-17 16:16:55 +00:00
PSchroedl 0564c80e07 Merge pull request #7 from pschroedl/revert-6-fix-install 2025-01-13 20:41:31 -08:00
John | Elite Encoder 2a59fe59b3 Revert "Fix node install from conda environments" 2025-01-13 23:10:45 -05:00
PSchroedl 8b00984581 Merge pull request #6 from eliteprox/fix-install 2025-01-09 19:41:11 -08:00
John | Elite Encoder 8db624cfe1 Update README.md 2025-01-09 21:49:26 -05:00
John | Elite Encoder 6258558746 Update README.md 2025-01-09 21:48:56 -05:00
John | Elite Encoder a4c053dae8 Update README.md 2025-01-09 21:34:35 -05:00
John | Elite Encoder 6b2c03f8bf Create pyproject.toml 2025-01-08 23:15:00 -05:00
John | Elite Encoder 0bbc1e0efa Create README.md
add install notes
2025-01-08 23:14:36 -05:00
John | Elite Encoder 60c60c5c2b remove self install from requirements.txt
Resolves issue with conda windows environments losing system environment variable context (e.g. CUDA_HOME error)
2025-01-08 22:56:46 -05:00
PSchroedl 4f587443fb Merge pull request #5 from pschroedl/unbreak_old_workflows_and_speedup
Unbreak old workflows and speedup
2024-12-09 23:58:21 -08:00
Peter Schroedl 37aa0d4c89 add small model option and config 2024-12-10 08:42:30 +01:00
Peter Schroedl 843ca3e733 reduce resolution internally to 512 2024-12-10 08:28:06 +01:00
Peter Schroedl f4e56bd733 make reset_tracking optional 2024-12-10 08:26:06 +01:00
PSchroedl de1fb0ab2a Merge pull request #4 from pschroedl/fix_point_coords
fix: update point coords scaling for mask
2024-12-09 21:47:43 -08:00
3 changed files with 125 additions and 6 deletions
+8 -5
View File
@@ -31,7 +31,7 @@ class DownloadAndLoadSAM2RealtimeModel:
def INPUT_TYPES(s):
return {"required": {
"model": ([
'sam2_hiera_tiny.pt',
'sam2_hiera_tiny.pt', 'sam2_hiera_small.pt',
],),
"segmentor": (
['realtime'],
@@ -70,7 +70,8 @@ class DownloadAndLoadSAM2RealtimeModel:
if not os.path.exists(model_path):
print(f"Downloading SAM2 model to: {model_path}")
url = "https://dl.fbaipublicfiles.com/segment_anything_2/072824/sam2_hiera_tiny.pt"
base_url = "https://dl.fbaipublicfiles.com/segment_anything_2/072824/"
url = f"{base_url}{model}"
response = requests.get(url, stream=True)
response.raise_for_status()
@@ -83,8 +84,9 @@ class DownloadAndLoadSAM2RealtimeModel:
config_dir = os.path.join(script_directory, "sam2_configs")
model_cfg = model.replace(".pt", ".yaml")
# Code ripped out of sam2.build_sam.build_sam2_camera_predictor to appease Hydra
model_cfg = "sam2_hiera_t.yaml" #TODO: remove hardcoded config and path
with initialize_config_dir(config_dir=config_dir, version_base=None):
cfg = compose(config_name=model_cfg)
@@ -140,12 +142,12 @@ class Sam2RealtimeSegmentation:
"required": {
"images": ("IMAGE",),
"sam2_model": ("SAM2MODEL",),
"reset_tracking": ("BOOLEAN", {"default": False}),
# "keep_model_loaded": ("BOOLEAN", {"default": True}),
},
"optional": {
"coordinates_positive": ("STRING", ),
"coordinates_negative": ("STRING", ),
"reset_tracking": ("BOOLEAN", {"default": False}),
# "bboxes": ("BBOX", ),
# "individual_objects": ("BOOLEAN", {"default": False}),
# "mask": ("MASK", ),
@@ -192,9 +194,9 @@ class Sam2RealtimeSegmentation:
images,
sam2_model,
# keep_model_loaded,
reset_tracking,
coordinates_positive=None,
coordinates_negative=None,
reset_tracking=False,
#point_labels=None,
# bboxes=None,
# individual_objects=False,
@@ -250,6 +252,7 @@ class Sam2RealtimeSegmentation:
# Create colored overlay for processed frames
mask_colored = torch.stack([mask] * 3, dim=2)
overlayed_frame = torch.add(frame * 0.7, mask_colored * 0.3)
processed_frames.append(overlayed_frame)
+116
View File
@@ -0,0 +1,116 @@
# @package _global_
# Model
model:
_target_: sam2_realtime.modeling.sam2_base.SAM2Base
image_encoder:
_target_: sam2_realtime.modeling.backbones.image_encoder.ImageEncoder
scalp: 1
trunk:
_target_: sam2_realtime.modeling.backbones.hieradet.Hiera
embed_dim: 96
num_heads: 1
stages: [1, 2, 11, 2]
global_att_blocks: [7, 10, 13]
window_pos_embed_bkg_spatial_size: [7, 7]
neck:
_target_: sam2_realtime.modeling.backbones.image_encoder.FpnNeck
position_encoding:
_target_: sam2_realtime.modeling.position_encoding.PositionEmbeddingSine
num_pos_feats: 256
normalize: true
scale: null
temperature: 10000
d_model: 256
backbone_channel_list: [768, 384, 192, 96]
fpn_top_down_levels: [2, 3] # output level 0 and 1 directly use the backbone features
fpn_interp_model: nearest
memory_attention:
_target_: sam2_realtime.modeling.memory_attention.MemoryAttention
d_model: 256
pos_enc_at_input: true
layer:
_target_: sam2_realtime.modeling.memory_attention.MemoryAttentionLayer
activation: relu
dim_feedforward: 2048
dropout: 0.1
pos_enc_at_attn: false
self_attention:
_target_: sam2_realtime.modeling.sam.transformer.RoPEAttention
rope_theta: 10000.0
feat_sizes: [32, 32]
embedding_dim: 256
num_heads: 1
downsample_rate: 1
dropout: 0.1
d_model: 256
pos_enc_at_cross_attn_keys: true
pos_enc_at_cross_attn_queries: false
cross_attention:
_target_: sam2_realtime.modeling.sam.transformer.RoPEAttention
rope_theta: 10000.0
feat_sizes: [32, 32]
rope_k_repeat: True
embedding_dim: 256
num_heads: 1
downsample_rate: 1
dropout: 0.1
kv_in_dim: 64
num_layers: 4
memory_encoder:
_target_: sam2_realtime.modeling.memory_encoder.MemoryEncoder
out_dim: 64
position_encoding:
_target_: sam2_realtime.modeling.position_encoding.PositionEmbeddingSine
num_pos_feats: 64
normalize: true
scale: null
temperature: 10000
mask_downsampler:
_target_: sam2_realtime.modeling.memory_encoder.MaskDownSampler
kernel_size: 3
stride: 2
padding: 1
fuser:
_target_: sam2_realtime.modeling.memory_encoder.Fuser
layer:
_target_: sam2_realtime.modeling.memory_encoder.CXBlock
dim: 256
kernel_size: 7
padding: 3
layer_scale_init_value: 1e-6
use_dwconv: True # depth-wise convs
num_layers: 2
num_maskmem: 7
image_size: 512
# apply scaled sigmoid on mask logits for memory encoder, and directly feed input mask as output mask
sigmoid_scale_for_mem_enc: 20.0
sigmoid_bias_for_mem_enc: -10.0
use_mask_input_as_output_without_sam: true
# Memory
directly_add_no_mem_embed: true
# use high-resolution feature map in the SAM mask decoder
use_high_res_features_in_sam: true
# output 3 masks on the first click on initial conditioning frames
multimask_output_in_sam: true
# SAM heads
iou_prediction_use_sigmoid: True
# cross-attend to object pointers from other frames (based on SAM output tokens) in the encoder
use_obj_ptrs_in_encoder: true
add_tpos_enc_to_obj_ptrs: false
only_obj_ptrs_in_the_past_for_eval: true
# object occlusion prediction
pred_obj_scores: true
pred_obj_scores_mlp: true
fixed_no_obj_ptr: true
# multimask tracking settings
multimask_output_for_tracking: true
use_multimask_token_for_obj_ptr: true
multimask_min_pt_num: 0
multimask_max_pt_num: 1
use_mlp_for_obj_ptr_proj: true
# Compilation flag
compile_image_encoder: False
@@ -85,7 +85,7 @@ model:
num_layers: 2
num_maskmem: 7
image_size: 1024
image_size: 512
# apply scaled sigmoid on mask logits for memory encoder, and directly feed input mask as output mask
# SAM decoder
sigmoid_scale_for_mem_enc: 20.0