4 Commits
Author SHA1 Message Date
Willie G d57a11d9a8 Add files via upload 2025-03-27 09:13:30 -04:00
Willie G c00bbab912 Add files via upload 2025-03-27 09:12:50 -04:00
Willie G 8f2a4d5e82 Delete LICENSE 2025-03-27 07:23:16 -04:00
Willie G 866c4acdde Initial commit 2025-03-27 07:13:57 -04:00
19 changed files with 1159 additions and 2175 deletions
-38
View File
@@ -1,38 +0,0 @@
---
name: Bug Report
about: Create a report to help us improve
title: ''
labels: ''
assignees: ''
---
<!--
🔍 STOP! Before creating a new issue:
1. Please search the closed issues first: https://github.com/ShmuelRonen/ComfyUI-LatentSyncWrapper/issues?q=is%3Aissue+is%3Aclosed
2. Check if your issue has already been resolved in past discussions.
3. If you find a similar issue, please add a comment there instead of creating a new one.
-->
**Describe the bug**
A clear and concise description of what the bug is.
**Steps To Reproduce**
1. Go to '...'
2. Click on '....'
3. Scroll down to '....'
4. See error
**Expected behavior**
A clear and concise description of what you expected to happen.
**Additional context**
Add any other context about the problem here.
A clear and concise description of what you expected to happen.
## Additional Context
Add any other context about the problem here (screenshots, logs, etc.).
## Environment
- OS: [e.g. Windows 10, Ubuntu 20.04]
- Browser: [e.g. Chrome, Firefox]
- Version: [e.g. 22]
-26
View File
@@ -1,26 +0,0 @@
name: Publish to Comfy registry
on:
workflow_dispatch:
push:
branches:
- main
- master
paths:
- "pyproject.toml"
jobs:
publish-node:
name: Publish Custom Node to registry
runs-on: ubuntu-latest
# if this is a forked repository. Skipping the workflow.
if: github.event.repository.fork == false
steps:
- name: Check out code
uses: actions/checkout@v4
with:
submodules: true
- name: Publish Custom Node
uses: Comfy-Org/publish-node-action@main
with:
## Add your own personal access token to your Github Repository secrets and reference it here.
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
-171
View File
@@ -1,171 +0,0 @@
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[cod]
*$py.class
# C extensions
*.so
# Distribution / packaging
.Python
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
share/python-wheels/
*.egg-info/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py,cover
.hypothesis/
.pytest_cache/
cover/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
# PyBuilder
.pybuilder/
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
# For a library or package, you might want to ignore these files since the code is
# intended to run in multiple environments; otherwise, check them in:
# .python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenv may install dependencies that don't work, or not
# install all needed dependencies.
#Pipfile.lock
# UV
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
#uv.lock
# poetry
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
#poetry.lock
# pdm
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
#pdm.lock
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
# in version control.
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
.pdm.toml
.pdm-python
.pdm-build/
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
__pypackages__/
# Celery stuff
celerybeat-schedule
celerybeat.pid
# SageMath parsed files
*.sage.py
# Environments
.env
.venv
env/
venv/
ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.mypy_cache/
.dmypy.json
dmypy.json
# Pyre type checker
.pyre/
# pytype static type analyzer
.pytype/
# Cython debug symbols
cython_debug/
# PyCharm
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
# and can be added to the global gitignore or merged into this file. For a more nuclear
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
#.idea/
# PyPI configuration file
.pypirc
+201 -201
View File
@@ -1,201 +1,201 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
+167 -229
View File
@@ -1,229 +1,167 @@
# (Out Dated) ComfyUI-Geeky-LatentSyncWrapper 1.5 (Mediapipe isn't compatible with some changes to ComfyUI, will need to downgrade python version of portable comfyUI and remove xformers).
Unofficial **optimized and enhanced** fork of [LatentSync 1.5](https://github.com/bytedance/LatentSync) implementation for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) on Windows and WSL 2.0.
Works with ComfyUI Portable version 3.49 https://github.com/comfyanonymous/ComfyUI/releases/tag/v0.3.49
This node provides advanced lip-sync capabilities in ComfyUI using ByteDance's LatentSync 1.5 model with **significantly improved performance, memory efficiency, and stability**. This fork focuses on speed, reliability, and conflict-free coexistence with other LatentSync implementations.
<img width="1893" height="1357" alt="workflow (4)" src="https://github.com/user-attachments/assets/51d3b98a-e549-4c71-a2f2-2af5cc52f291" />
## Why This Fork? Performance & Stability
**🚀 Much Faster Performance**: This implementation is significantly faster than other versions and eliminates OOM (Out of Memory) errors that plague other implementations.
**🧠 Better Memory Management**: Intelligent VRAM usage with user-selectable settings (high/medium/low) and automatic cleanup prevents memory issues.
**🔒 Conflict-Free**: Can be installed alongside other LatentSync implementations without interference - uses isolated paths and unique node names.
**⚡ LatentSync 1.5 vs 1.6**: We use LatentSync 1.5 instead of 1.6 because:
- **More Stable**: 1.5 has proven stability and reliability in production use
- **Better Performance**: 1.5 runs faster and uses less VRAM than 1.6
- **No Manual Downloads**: 1.5 models download automatically, unlike 1.6's private repository requirements
- **Fewer Dependencies**: Simpler, more reliable dependency chain
## What's New in This Optimized Fork?
### Performance Enhancements
1. **Advanced Memory Management**: Intelligent VRAM allocation with user-selectable modes
2. **Faster Processing**: Optimized batch processing and GPU utilization
3. **No OOM Errors**: Comprehensive memory cleanup and management
4. **Mixed Precision Support**: Automatic FP16 optimization when beneficial
### User Experience Improvements
5. **Single Image Support**: Process individual images with LatentSync
6. **Batch Image Processing**: Process multiple images efficiently
7. **Smart Temp Management**: Isolated temporary directories prevent conflicts
8. **Better Error Handling**: Robust error recovery and informative messages
### Compatibility Features
9. **Conflict-Free Installation**: Can coexist with ShmuelRonen's implementation
10. **Unique Node Names**: "Geeky" prefixed nodes prevent naming conflicts
11. **Isolated Model Storage**: Uses `geeky_checkpoints/` directory
12. **Automatic Path Management**: Handles compatibility transparently
## Original LatentSync 1.5 Features
1. **Temporal Layer Improvements**: Corrected implementation provides significantly improved temporal consistency compared to version 1.0
2. **Better Chinese Language Support**: Performance on Chinese videos is substantially improved through additional training data
3. **Reduced VRAM Requirements**: Optimized to run on 20GB VRAM (RTX 3090 compatible) through various optimizations:
- Gradient checkpointing in U-Net, VAE, SyncNet and VideoMAE
- Native PyTorch FlashAttention-2 implementation (no xFormers dependency)
- More efficient CUDA cache management
- Focused training of temporal and audio cross-attention layers only
4. **Code Optimizations**:
- Removed dependencies on xFormers and Triton
- Upgraded to diffusers 0.32.2
## Compatibility with Other LatentSync Nodes
This repository can be installed alongside ShmuelRonen's ComfyUI-LatentSyncWrapper **without conflicts**:
- ✅ **Different node names**: Geeky nodes use "Geeky" prefix ("Geeky LatentSync 1.5 (Optimized)")
- ✅ **Separate checkpoints**: Uses `geeky_checkpoints/` directory
- ✅ **Independent models**: Downloads to isolated paths
- ✅ **No shared resources**: Completely separate from other LatentSync implementations
- ✅ **Isolated temp directories**: Prevents interference with other nodes
Both repositories can coexist and users can choose which nodes to use based on their performance needs.
## Prerequisites
Before installing this node, you must install the following in order:
1. [ComfyUI](https://github.com/comfyanonymous/ComfyUI) installed and working
2. FFmpeg installed on your system:
- Windows: Download from [here](https://github.com/BtbN/FFmpeg-Builds/releases) and add to system PATH
## Installation
Only proceed with installation after confirming all prerequisites are installed and working.
1. Clone this repository into your ComfyUI custom_nodes directory:
```bash
cd ComfyUI/custom_nodes
git clone https://github.com/GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper.git
cd ComfyUI-Geeky-LatentSyncWrapper
pip install -r requirements.txt
```
## Required Dependencies
```
diffusers>=0.32.2
transformers
huggingface-hub
omegaconf
einops
opencv-python
mediapipe
face-alignment
decord
ffmpeg-python
safetensors
soundfile
```
## Note on Model Downloads
On first use, the node will **automatically download** required model files from HuggingFace:
- LatentSync 1.5 UNet model (~5GB)
- Whisper model for audio processing (~1.6GB)
- All models download automatically - no manual intervention required
- Models are stored in isolated `geeky_checkpoints/` directory
### Checkpoint Directory Structure
After successful installation and model download, your checkpoint directory structure will look like this:
```
./geeky_checkpoints/
|-- .cache/
|-- auxiliary/
|-- whisper/
| `-- tiny.pt
|-- config.json
|-- latentsync_unet.pt (~5GB)
|-- stable_syncnet.pt (~1.6GB)
```
Make sure all these files are present for proper functionality. The main model files are:
- `latentsync_unet.pt`: The primary LatentSync 1.5 model
- `stable_syncnet.pt`: The SyncNet model for lip-sync supervision
- `whisper/tiny.pt`: The Whisper model for audio processing
## Usage
### For Videos:
1. Select an input video file with a video loader
2. Load an audio file using ComfyUI audio loader
3. (Optional) Set a seed value for reproducible results
4. (Optional) Adjust the lips_expression parameter to control lip movement intensity
5. (Optional) Modify the inference_steps parameter to balance quality and speed
6. (Optional) Choose VRAM usage setting based on your GPU
7. Connect to the **Geeky LatentSync 1.5 (Optimized)** node
8. Run the workflow
### For Single Images:
1. Load a single image using ComfyUI's image loader
2. Load an audio file using ComfyUI audio loader
3. Connect to the **Geeky LatentSync 1.5 (Optimized)** node
4. Adjust parameters as needed
5. Run the workflow
### For Batch Images:
1. Load multiple images using ComfyUI's batch image loader or image list to batch node
2. Load an audio file using ComfyUI audio loader
3. Connect to the **Geeky LatentSync 1.5 (Optimized)** node
4. Adjust parameters as needed
5. Run the workflow
The processed video or images will be saved in ComfyUI's output directory.
### Node Parameters:
- `images`: Input image(s) - supports single images, video frames, or batch processing
- `audio`: Audio input from ComfyUI audio loader
- `seed`: Random seed for reproducible results (default: 1247)
- `lips_expression`: Controls the expressiveness of lip movements (default: 1.5)
- Higher values (2.0-3.0): More pronounced lip movements, better for expressive speech
- Lower values (1.0-1.5): Subtler lip movements, better for calm speech
- This parameter affects the model's guidance scale, balancing between natural movement and lip sync accuracy
- `inference_steps`: Number of denoising steps during inference (default: 20)
- Higher values (30-50): Better quality results but slower processing
- Lower values (10-15): Faster processing but potentially lower quality
- The default of 20 usually provides a good balance between quality and speed
- `vram_usage`: **NEW** - Choose memory usage profile (default: medium)
- **High**: Maximum performance, uses 95% VRAM, enables all optimizations
- **Medium**: Balanced performance, uses 85% VRAM, good for most users
- **Low**: Conservative usage, uses 75% VRAM, for systems with limited memory
### Available Nodes:
- **Geeky LatentSync 1.5 (Optimized)**: Main lip-sync processing node
- **Geeky Video Length Adjuster (Fast)**: Utility node for video/audio length matching
### Tips for Better Results:
- **Performance**: Start with "medium" VRAM usage and increase to "high" if you have sufficient GPU memory
- **Quality**: For speeches or presentations, try increasing lips_expression to 2.0-2.5
- **Efficiency**: For quick previews, use "low" VRAM setting with 10-15 inference steps
- **Stability**: This implementation handles single images automatically by duplicating frames to match audio length
- **Memory**: The optimized memory management prevents OOM errors even with long audio clips
## Performance Comparison
| Feature | This Fork (Geeky) | Original Implementation |
|---------|-------------------|------------------------|
| OOM Errors | ❌ None | ✅ Frequent |
| Processing Speed | 🚀 Much Faster | 🐌 Slower |
| Memory Usage | 🧠 Optimized | 💾 High |
| VRAM Settings | ✅ 3 Modes | ❌ Fixed |
| Conflict-Free | ✅ Yes | ❌ No |
| Auto Downloads | ✅ Yes | ⚠️ Manual (1.6) |
## Known Limitations
- Works best with clear, frontal face images/videos
- Currently does not support anime/cartoon faces
- Video should be at 25 FPS (will be automatically converted)
- Face should be visible throughout the image/video
- Single images are automatically extended to match audio duration
## Troubleshooting
### Common Issues:
1. **"Geeky model checkpoints already exist"**: This is normal - models are cached for faster startup
2. **Memory errors**: Try lowering VRAM usage setting from high → medium → low
3. **Slow performance**: Ensure you're using a CUDA-compatible GPU and try "high" VRAM setting
4. **Node not appearing**: Restart ComfyUI after installation and refresh your browser
## Credits
This optimized fork is based on:
- [LatentSync 1.5](https://github.com/bytedance/LatentSync) by ByteDance Research
- [ComfyUI-LatentSyncWrapper](https://github.com/ShmuelRonen/ComfyUI-LatentSyncWrapper) by ShmuelRonen
- [ComfyUI](https://github.com/comfyanonymous/ComfyUI)
Special thanks to the original developers for their groundbreaking work. This fork focuses on performance optimization, memory efficiency, and user experience improvements.
## License
This project is licensed under the Apache License 2.0 - see the LICENSE file for details.
# ComfyUI-Geeky-LatentSyncWrapper 1.5
Unofficial enhanced fork of [LatentSync 1.5](https://github.com/bytedance/LatentSync) implementation for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) on Windows and WSL 2.0.
This node provides advanced lip-sync capabilities in ComfyUI using ByteDance's LatentSync 1.5 model. It allows you to synchronize video lips with audio input with improved temporal consistency and better performance on a wider range of languages. This fork adds support for both single images and batch image processing.
<img width="883" alt="Screenshot 2025-03-27 082328" src="https://github.com/user-attachments/assets/9cb30dd5-3507-4565-a917-ae0ede1a2e89" />
<img width="748" alt="Screenshot 2025-03-27 082535" src="https://github.com/user-attachments/assets/3fb7a39e-da9e-444c-a43a-15de48fa57a9" />
## What's new in this fork?
1. **Single Image Support**: Process individual images with LatentSync
2. **Batch Image Processing**: Process multiple images in a batch for efficient workflows
3. **All original LatentSync 1.5 features**: Enhanced temporal consistency, better language support, and reduced VRAM requirements
## Original LatentSync 1.5 Features
1. **Temporal Layer Improvements**: Corrected implementation now provides significantly improved temporal consistency compared to version 1.0
2. **Better Chinese Language Support**: Performance on Chinese videos is now substantially improved through additional training data
3. **Reduced VRAM Requirements**: Now only requires 20GB VRAM (can run on RTX 3090) through various optimizations:
- Gradient checkpointing in U-Net, VAE, SyncNet and VideoMAE
- Native PyTorch FlashAttention-2 implementation (no xFormers dependency)
- More efficient CUDA cache management
- Focused training of temporal and audio cross-attention layers only
4. **Code Optimizations**:
- Removed dependencies on xFormers and Triton
- Upgraded to diffusers 0.32.2
## Prerequisites
Before installing this node, you must install the following in order:
1. [ComfyUI](https://github.com/comfyanonymous/ComfyUI) installed and working
2. FFmpeg installed on your system:
- Windows: Download from [here](https://github.com/BtbN/FFmpeg-Builds/releases) and add to system PATH
## Installation
Only proceed with installation after confirming all prerequisites are installed and working.
1. Clone this repository into your ComfyUI custom_nodes directory:
```bash
cd ComfyUI/custom_nodes
git clone https://github.com/GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper.git
cd ComfyUI-Geeky-LatentSyncWrapper
pip install -r requirements.txt
```
## Required Dependencies
```
diffusers>=0.32.2
transformers
huggingface-hub
omegaconf
einops
opencv-python
mediapipe
face-alignment
decord
ffmpeg-python
safetensors
soundfile
```
## Note on Model Downloads
On first use, the node will automatically download required model files from HuggingFace:
- LatentSync 1.5 UNet model
- Whisper model for audio processing
- You can also manually download the models from HuggingFace repo: https://huggingface.co/ByteDance/LatentSync-1.5
### Checkpoint Directory Structure
After successful installation and model download, your checkpoint directory structure should look like this:
```
./checkpoints/
|-- .cache/
|-- auxiliary/
|-- whisper/
| `-- tiny.pt
|-- config.json
|-- latentsync_unet.pt (~5GB)
|-- stable_syncnet.pt (~1.6GB)
```
Make sure all these files are present for proper functionality. The main model files are:
- `latentsync_unet.pt`: The primary LatentSync 1.5 model
- `stable_syncnet.pt`: The SyncNet model for lip-sync supervision
- `whisper/tiny.pt`: The Whisper model for audio processing
## Usage
### For Videos:
1. Select an input video file with AceNodes video loader
2. Load an audio file using ComfyUI audio loader
3. (Optional) Set a seed value for reproducible results
4. (Optional) Adjust the lips_expression parameter to control lip movement intensity
5. (Optional) Modify the inference_steps parameter to balance quality and speed
6. Connect to the LatentSync1.5 node
7. Run the workflow
### For Single Images:
1. Load a single image using ComfyUI's image loader
2. Load an audio file using ComfyUI audio loader
3. Connect to the LatentSync1.5 node with the single image mode enabled
4. Adjust parameters as needed
5. Run the workflow
### For Batch Images:
1. Load multiple images using ComfyUI's batch image loader or image list to batch node
2. Load an audio file using ComfyUI audio loader
3. Connect to the LatentSync1.5 node with the batch processing mode enabled
4. Adjust parameters as needed
5. Run the workflow
The processed video or images will be saved in ComfyUI's output directory.
### Node Parameters:
- `input_type`: Select between video, single image, or batch images
- `video_path`: Path to input video file (for video mode)
- `image`: Input single image (for single image mode)
- `image_batch`: Input batch of images (for batch image mode)
- `audio`: Audio input from ComfyUI audio loader
- `seed`: Random seed for reproducible results (default: 1247)
- `lips_expression`: Controls the expressiveness of lip movements (default: 1.5)
- Higher values (2.0-3.0): More pronounced lip movements, better for expressive speech
- Lower values (1.0-1.5): Subtler lip movements, better for calm speech
- This parameter affects the model's guidance scale, balancing between natural movement and lip sync accuracy
- `inference_steps`: Number of denoising steps during inference (default: 20)
- Higher values (30-50): Better quality results but slower processing
- Lower values (10-15): Faster processing but potentially lower quality
- The default of 20 usually provides a good balance between quality and speed
- `batch_size`: Number of images to process at once in batch mode (default: 4)
- Higher values may require more VRAM
### Tips for Better Results:
- For speeches or presentations where clear lip movements are important, try increasing the lips_expression value to 2.0-2.5
- For casual conversations, the default value of 1.5 usually works well
- If lip movements appear unnatural or exaggerated, try lowering the lips_expression value
- Different values may work better for different languages and speech patterns
- If you need higher quality results and have time to wait, increase inference_steps to 30-50
- For quicker previews or less critical applications, reduce inference_steps to 10-15
- When processing batch images, adjust batch_size based on your available VRAM
## Known Limitations
- Works best with clear, frontal face images/videos
- Currently does not support anime/cartoon faces
- Video should be at 25 FPS (will be automatically converted)
- Face should be visible throughout the image/video
- Batch processing may require significant VRAM depending on batch size
## Credits
This fork is based on:
- [LatentSync 1.5](https://github.com/bytedance/LatentSync) by ByteDance Research
- [ComfyUI-LatentSyncWrapper](https://github.com/ShmuelRonen/ComfyUI-LatentSyncWrapper) by ShmuelRonen
- [ComfyUI](https://github.com/comfyanonymous/ComfyUI)
## License
This project is licensed under the Apache License 2.0 - see the LICENSE file for details.
+2 -2
View File
@@ -1,3 +1,3 @@
from .nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
from .nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS']
+92 -288
View File
@@ -3,25 +3,6 @@ import tempfile
import uuid
import sys
import shutil
import time
# Global model cache to avoid reloading models - Using unique name to avoid conflicts
_GEEKY_MODEL_CACHE = {}
# Function to check for potential conflicts with other LatentSync implementations
def check_for_conflicts():
"""Check if other LatentSync implementations might conflict"""
try:
import folder_paths
custom_nodes_dir = folder_paths.get_folder_paths("custom_nodes")[0]
original_path = os.path.join(custom_nodes_dir, "ComfyUI-LatentSyncWrapper")
if os.path.exists(original_path):
print("[Geeky LatentSync] Detected ComfyUI-LatentSyncWrapper - using isolated paths to avoid conflicts")
return True
except:
pass
return False
# Function to find ComfyUI directories
def get_comfyui_temp_dir():
@@ -90,10 +71,6 @@ def cleanup_comfyui_temp_directories():
except Exception as e:
print(f"Error cleaning up temp directories: {str(e)}")
def get_unique_temp_path(suffix=""):
"""Generate unique temp paths for geeky wrapper"""
return os.path.join(tempfile.gettempdir(), f"geeky_latentsync_{uuid.uuid4().hex[:8]}{suffix}")
# Create a module-level function to set up system-wide temp directory
def init_temp_directories():
"""Initialize global temporary directory settings"""
@@ -103,19 +80,9 @@ def init_temp_directories():
# Generate a unique base directory for this module
system_temp = tempfile.gettempdir()
unique_id = str(uuid.uuid4())[:8]
temp_base_path = os.path.join(system_temp, f"geeky_latentsync_{unique_id}")
temp_base_path = os.path.join(system_temp, f"latentsync_{unique_id}")
os.makedirs(temp_base_path, exist_ok=True)
# Create a persistent model cache directory
model_cache_dir = os.path.join(system_temp, "geeky_latentsync_model_cache")
os.makedirs(model_cache_dir, exist_ok=True)
# Link it into our temp directory for convenience
try:
os.symlink(model_cache_dir, os.path.join(temp_base_path, "model_cache"), target_is_directory=True)
except (OSError, NotImplementedError):
# If symlinks aren't supported, just use the cache dir directly
shutil.copytree(model_cache_dir, os.path.join(temp_base_path, "model_cache"), dirs_exist_ok=True)
# Override environment variables that control temp directories
os.environ['TMPDIR'] = temp_base_path
os.environ['TEMP'] = temp_base_path
@@ -151,25 +118,13 @@ def init_temp_directories():
# Function to clean up everything when the module exits
def module_cleanup():
"""Clean up all resources when the module is unloaded"""
global MODULE_TEMP_DIR, _GEEKY_MODEL_CACHE
global MODULE_TEMP_DIR
# Clear model cache references to free memory
_GEEKY_MODEL_CACHE.clear()
# Clean up temp directory except model cache
# Clean up our module temp directory
if MODULE_TEMP_DIR and os.path.exists(MODULE_TEMP_DIR):
try:
for item in os.listdir(MODULE_TEMP_DIR):
if item != "model_cache":
path = os.path.join(MODULE_TEMP_DIR, item)
if os.path.isdir(path):
shutil.rmtree(path, ignore_errors=True)
else:
try:
os.remove(path)
except:
pass
print(f"Cleaned up module temp directory (preserving model cache)")
shutil.rmtree(MODULE_TEMP_DIR, ignore_errors=True)
print(f"Cleaned up module temp directory: {MODULE_TEMP_DIR}")
except:
pass
@@ -200,9 +155,6 @@ from PIL import Image
from decimal import Decimal, ROUND_UP
import requests
# Check for conflicts with other implementations
conflict_detected = check_for_conflicts()
# Modify folder_paths module to use our temp directory
if hasattr(folder_paths, "get_temp_directory"):
original_get_temp = folder_paths.get_temp_directory
@@ -211,44 +163,12 @@ else:
# Add the function if it doesn't exist
setattr(folder_paths, 'get_temp_directory', lambda: MODULE_TEMP_DIR)
def get_cached_model(model_path, model_type, device):
"""Load model from geeky-specific cache or disk and cache it"""
global _GEEKY_MODEL_CACHE
cache_key = f"geeky_{model_type}_{model_path}"
if cache_key in _GEEKY_MODEL_CACHE:
# Check if the cached model is on the right device
cached_model = _GEEKY_MODEL_CACHE[cache_key]
model_device = next(cached_model.parameters()).device
if str(model_device) == str(device):
print(f"Using cached {model_type} model from Geeky cache")
return cached_model
else:
print(f"Moving cached {model_type} model to {device}")
cached_model = cached_model.to(device)
return cached_model
print(f"Loading {model_type} model from disk into Geeky cache")
# Load the model
model = torch.load(model_path, map_location=device)
# Cache the model
_GEEKY_MODEL_CACHE[cache_key] = model
return model
def import_inference_script(script_path):
"""Import a Python file as a module using its file path."""
if not os.path.exists(script_path):
raise ImportError(f"Script not found: {script_path}")
module_name = "geeky_latentsync_inference" # Unique module name
# Check if the module is already loaded
if module_name in sys.modules:
print("Using previously imported Geeky inference module")
return sys.modules[module_name]
print(f"Importing Geeky inference script from {script_path}")
module_name = "latentsync_inference"
spec = importlib.util.spec_from_file_location(module_name, script_path)
if spec is None:
raise ImportError(f"Failed to create module spec for {script_path}")
@@ -294,6 +214,7 @@ def check_ffmpeg():
def check_and_install_dependencies():
if not check_ffmpeg():
raise RuntimeError("FFmpeg is required but not found")
required_packages = [
'omegaconf',
'transformers',
@@ -303,20 +224,10 @@ def check_and_install_dependencies():
'diffusers',
'ffmpeg-python'
]
# Check if we've already run this function successfully
cache_dir = os.path.join(MODULE_TEMP_DIR, "model_cache")
# Create the cache directory if it doesn't exist
os.makedirs(cache_dir, exist_ok=True)
cache_marker = os.path.join(cache_dir, ".geeky_deps_installed")
if os.path.exists(cache_marker):
print("Geeky dependencies already verified, skipping check.")
return
def is_package_installed(package_name):
return importlib.util.find_spec(package_name) is not None
def install_package(package):
python_exe = sys.executable
try:
@@ -327,29 +238,15 @@ def check_and_install_dependencies():
except subprocess.CalledProcessError as e:
print(f"Error installing {package}: {str(e)}")
raise RuntimeError(f"Failed to install required package: {package}")
missing_packages = []
for package in required_packages:
if not is_package_installed(package):
missing_packages.append(package)
if missing_packages:
print(f"Installing missing packages: {', '.join(missing_packages)}")
for package in missing_packages:
print(f"Installing required package: {package}")
try:
install_package(package)
except Exception as e:
print(f"Warning: Failed to install {package}: {str(e)}")
raise
else:
print("All required packages are already installed.")
# Create marker file
try:
with open(cache_marker, 'w') as f:
f.write(f"Geeky dependencies checked on {time.ctime()}")
except Exception as e:
print(f"Warning: Could not create cache marker file: {str(e)}")
def normalize_path(path):
"""Normalize path to handle spaces and special characters"""
@@ -380,24 +277,10 @@ def get_ext_dir(subpath=None, mkdir=False):
def download_model(url, save_path):
"""Download a model from a URL and save it to the specified path."""
os.makedirs(os.path.dirname(save_path), exist_ok=True)
# Check if file already exists
if os.path.exists(save_path):
print(f"Model file already exists at {save_path}, skipping download.")
return
print(f"Downloading {url} to {save_path}...")
response = requests.get(url, stream=True)
total_size = int(response.headers.get('content-length', 0))
downloaded = 0
with open(save_path, "wb") as f:
for chunk in response.iter_content(chunk_size=8192):
f.write(chunk)
downloaded += len(chunk)
if total_size > 0:
percent = (downloaded / total_size) * 100
print(f"\rDownload progress: {percent:.1f}%", end="")
print("\nDownload complete")
def pre_download_models():
"""Pre-download all required models."""
@@ -409,23 +292,13 @@ def pre_download_models():
cache_dir = os.path.join(MODULE_TEMP_DIR, "model_cache")
os.makedirs(cache_dir, exist_ok=True)
# Check if we've already run this function successfully by creating a marker file
cache_marker = os.path.join(cache_dir, ".geeky_cache_complete")
if os.path.exists(cache_marker):
print("Pre-downloaded Geeky models already exist, skipping download.")
return
for model_name, url in models.items():
save_path = os.path.join(cache_dir, model_name)
if not os.path.exists(save_path):
print(f"Downloading {model_name}...")
download_model(url, save_path)
else:
print(f"{model_name} already exists in Geeky cache.")
# Create marker file to indicate successful completion
with open(cache_marker, 'w') as f:
f.write(f"Geeky cache completed on {time.ctime()}")
print(f"{model_name} already exists in cache.")
def setup_models():
"""Setup and pre-download all required models."""
@@ -435,9 +308,9 @@ def setup_models():
# Pre-download additional models
pre_download_models()
# Existing setup logic for LatentSync models - using unique directory
# Existing setup logic for LatentSync models
cur_dir = get_ext_dir()
ckpt_dir = os.path.join(cur_dir, "geeky_checkpoints") # Changed from "checkpoints" to "geeky_checkpoints"
ckpt_dir = os.path.join(cur_dir, "checkpoints")
whisper_dir = os.path.join(ckpt_dir, "whisper")
os.makedirs(ckpt_dir, exist_ok=True)
os.makedirs(whisper_dir, exist_ok=True)
@@ -449,28 +322,24 @@ def setup_models():
unet_path = os.path.join(ckpt_dir, "latentsync_unet.pt")
whisper_path = os.path.join(whisper_dir, "tiny.pt")
# Only download if the files don't already exist
if os.path.exists(unet_path) and os.path.exists(whisper_path):
print("Geeky model checkpoints already exist, skipping download.")
return
print("Downloading required Geeky model checkpoints... This may take a while.")
try:
from huggingface_hub import snapshot_download
snapshot_download(repo_id="ByteDance/LatentSync-1.5",
allow_patterns=["latentsync_unet.pt", "whisper/tiny.pt"],
local_dir=ckpt_dir,
local_dir_use_symlinks=False,
cache_dir=temp_downloads)
print("Geeky model checkpoints downloaded successfully!")
except Exception as e:
print(f"Error downloading Geeky models: {str(e)}")
print("\nPlease download models manually for Geeky LatentSync:")
print("1. Visit: https://huggingface.co/chunyu-li/LatentSync")
print("2. Download: latentsync_unet.pt and whisper/tiny.pt")
print(f"3. Place them in: {ckpt_dir}")
print(f" with whisper/tiny.pt in: {whisper_dir}")
raise RuntimeError("Geeky model download failed. See instructions above.")
if not (os.path.exists(unet_path) and os.path.exists(whisper_path)):
print("Downloading required model checkpoints... This may take a while.")
try:
from huggingface_hub import snapshot_download
snapshot_download(repo_id="ByteDance/LatentSync-1.5",
allow_patterns=["latentsync_unet.pt", "whisper/tiny.pt"],
local_dir=ckpt_dir,
local_dir_use_symlinks=False,
cache_dir=temp_downloads)
print("Model checkpoints downloaded successfully!")
except Exception as e:
print(f"Error downloading models: {str(e)}")
print("\nPlease download models manually:")
print("1. Visit: https://huggingface.co/chunyu-li/LatentSync")
print("2. Download: latentsync_unet.pt and whisper/tiny.pt")
print(f"3. Place them in: {ckpt_dir}")
print(f" with whisper/tiny.pt in: {whisper_dir}")
raise RuntimeError("Model download failed. See instructions above.")
class GeekyLatentSyncNode:
def __init__(self):
@@ -480,8 +349,8 @@ class GeekyLatentSyncNode:
os.makedirs(MODULE_TEMP_DIR, exist_ok=True)
# Ensure ComfyUI temp doesn't exist
comfyui_temp = get_comfyui_temp_dir()
if comfyui_temp and os.path.exists(comfyui_temp):
comfyui_temp = "D:\\ComfyUI_windows\\temp"
if os.path.exists(comfyui_temp):
backup_name = f"{comfyui_temp}_backup_{uuid.uuid4().hex[:8]}"
try:
os.rename(comfyui_temp, backup_name)
@@ -499,7 +368,6 @@ class GeekyLatentSyncNode:
"seed": ("INT", {"default": 1247}),
"lips_expression": ("FLOAT", {"default": 1.5, "min": 1.0, "max": 3.0, "step": 0.1}),
"inference_steps": ("INT", {"default": 20, "min": 1, "max": 999, "step": 1}),
"vram_usage": (["high", "medium", "low"], {"default": "medium"}),
},}
CATEGORY = "GeekyLatentSync"
@@ -519,66 +387,51 @@ class GeekyLatentSyncNode:
processed_batch = processed_batch[..., :3]
return processed_batch
def inference(self, images, audio, seed, lips_expression=1.5, inference_steps=20, vram_usage="medium"):
# Add timing information
import time
start_time = time.time()
def inference(self, images, audio, seed, lips_expression=1.5, inference_steps=20):
# Use our module temp directory
global MODULE_TEMP_DIR
# Define timing checkpoint function
def log_timing(step):
elapsed = time.time() - start_time
print(f"[Geeky {elapsed:.2f}s] {step}")
log_timing("Starting Geeky inference")
# Get GPU capabilities and memory
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
BATCH_SIZE = 4
use_mixed_precision = False
# Set VRAM usage based on user preference
if torch.cuda.is_available():
gpu_mem = torch.cuda.get_device_properties(0).total_memory
# Convert to GB
gpu_mem_gb = gpu_mem / (1024 ** 3)
# Dynamic batch size and settings based on VRAM usage preference
if vram_usage == "high":
BATCH_SIZE = min(32, 120 // inference_steps)
# Dynamically adjust batch size based on GPU memory
if gpu_mem_gb > 20: # High-end GPUs
BATCH_SIZE = 32
enable_tf32 = True
use_mixed_precision = True
torch.backends.cudnn.benchmark = True
torch.backends.cuda.matmul.allow_tf32 = True if hasattr(torch.backends.cuda, "matmul") else False
torch.backends.cudnn.allow_tf32 = True
torch.cuda.set_per_process_memory_fraction(0.95)
print(f"Using Geeky high VRAM settings with {BATCH_SIZE} batch size")
elif vram_usage == "medium":
BATCH_SIZE = min(16, 80 // inference_steps)
elif gpu_mem_gb > 8: # Mid-range GPUs
BATCH_SIZE = 16
enable_tf32 = False
use_mixed_precision = True
torch.backends.cudnn.benchmark = True
torch.cuda.set_per_process_memory_fraction(0.85)
print(f"Using Geeky medium VRAM settings with {BATCH_SIZE} batch size")
else: # low
BATCH_SIZE = min(8, 40 // inference_steps)
else: # Lower-end GPUs
BATCH_SIZE = 8
enable_tf32 = False
use_mixed_precision = False
torch.cuda.set_per_process_memory_fraction(0.75)
print(f"Using Geeky low VRAM settings with {BATCH_SIZE} batch size")
# Set performance options based on GPU capability
torch.backends.cudnn.benchmark = True
if enable_tf32:
torch.backends.cuda.matmul.allow_tf32 = True
torch.backends.cudnn.allow_tf32 = True
# Clear GPU cache before processing
torch.cuda.empty_cache()
else:
# CPU fallback settings
BATCH_SIZE = 4
print("No GPU detected, using CPU with minimal batch size")
torch.cuda.set_per_process_memory_fraction(0.8)
# Create a run-specific subdirectory in our temp directory
run_id = ''.join(random.choice("abcdefghijklmnopqrstuvwxyz") for _ in range(5))
temp_dir = os.path.join(MODULE_TEMP_DIR, f"geeky_run_{run_id}")
temp_dir = os.path.join(MODULE_TEMP_DIR, f"run_{run_id}")
os.makedirs(temp_dir, exist_ok=True)
# Ensure ComfyUI temp doesn't exist again
comfyui_temp = get_comfyui_temp_dir()
if comfyui_temp and os.path.exists(comfyui_temp):
# Ensure ComfyUI temp doesn't exist again (in case something recreated it)
comfyui_temp = "D:\\ComfyUI_windows\\temp"
if os.path.exists(comfyui_temp):
backup_name = f"{comfyui_temp}_backup_{uuid.uuid4().hex[:8]}"
try:
os.rename(comfyui_temp, backup_name)
@@ -591,14 +444,13 @@ class GeekyLatentSyncNode:
try:
# Create temporary file paths in our system temp directory
temp_video_path = os.path.join(temp_dir, f"geeky_temp_{run_id}.mp4")
output_video_path = os.path.join(temp_dir, f"geeky_latentsync_{run_id}_out.mp4")
audio_path = os.path.join(temp_dir, f"geeky_latentsync_{run_id}_audio.wav")
temp_video_path = os.path.join(temp_dir, f"temp_{run_id}.mp4")
output_video_path = os.path.join(temp_dir, f"latentsync_{run_id}_out.mp4")
audio_path = os.path.join(temp_dir, f"latentsync_{run_id}_audio.wav")
# Get the extension directory
cur_dir = os.path.dirname(os.path.abspath(__file__))
log_timing("Processing input frames")
# Process input frames
if isinstance(images, list):
frames = torch.stack(images).to(device)
@@ -634,9 +486,8 @@ class GeekyLatentSyncNode:
single_frame = frames[0]
duplicated_frames = single_frame.unsqueeze(0).repeat(required_frames, 1, 1, 1)
frames = duplicated_frames
print(f"Geeky: Duplicated single image to create {required_frames} frames matching audio duration")
print(f"Duplicated single image to create {required_frames} frames matching audio duration")
log_timing("Processing audio")
# Resample audio if needed
if sample_rate != 16000:
new_sample_rate = 16000
@@ -653,7 +504,6 @@ class GeekyLatentSyncNode:
"sample_rate": sample_rate
}
log_timing("Saving temporary files")
# Move waveform to CPU for saving
waveform_cpu = waveform.cpu()
torchaudio.save(audio_path, waveform_cpu, sample_rate)
@@ -678,19 +528,13 @@ class GeekyLatentSyncNode:
packet = stream.encode(None)
container.mux(packet)
container.close()
# Free up memory after saving
del frames_cpu, waveform_cpu
if torch.cuda.is_available():
torch.cuda.empty_cache()
log_timing("Setting up model paths")
# Define paths to required files and configs - using geeky_checkpoints
# Define paths to required files and configs
inference_script_path = os.path.join(cur_dir, "scripts", "inference.py")
config_path = os.path.join(cur_dir, "configs", "unet", "stage2.yaml")
scheduler_config_path = os.path.join(cur_dir, "configs")
ckpt_path = os.path.join(cur_dir, "geeky_checkpoints", "latentsync_unet.pt") # Updated path
whisper_ckpt_path = os.path.join(cur_dir, "geeky_checkpoints", "whisper", "tiny.pt") # Updated path
ckpt_path = os.path.join(cur_dir, "checkpoints", "latentsync_unet.pt")
whisper_ckpt_path = os.path.join(cur_dir, "checkpoints", "whisper", "tiny.pt")
# Create config and args
config = OmegaConf.load(config_path)
@@ -728,25 +572,6 @@ class GeekyLatentSyncNode:
mask_image_path=mask_image_path
)
# CRITICAL FIX: Create symlink or copy to handle hardcoded paths in inference script
old_checkpoints_dir = os.path.join(cur_dir, "checkpoints")
if not os.path.exists(old_checkpoints_dir):
try:
# Try to create a symlink first (faster)
os.symlink(os.path.join(cur_dir, "geeky_checkpoints"), old_checkpoints_dir)
print(f"Created symlink from {old_checkpoints_dir} to geeky_checkpoints")
except (OSError, NotImplementedError):
# If symlinks aren't supported, copy the directory
shutil.copytree(os.path.join(cur_dir, "geeky_checkpoints"), old_checkpoints_dir, dirs_exist_ok=True)
# Create a marker file so we know this is our temporary copy
with open(os.path.join(old_checkpoints_dir, ".geeky_temp_copy"), 'w') as f:
f.write("Temporary copy created by Geeky LatentSync")
print(f"Copied geeky_checkpoints to {old_checkpoints_dir}")
elif os.path.isdir(old_checkpoints_dir) and not os.path.islink(old_checkpoints_dir):
# If checkpoints directory exists and is not our symlink, warn user
print("Warning: Found existing 'checkpoints' directory. This might conflict with other LatentSync implementations.")
print("Geeky LatentSync will use the existing directory but recommend using separate installations.")
# Set PYTHONPATH to include our directories
package_root = os.path.dirname(cur_dir)
if package_root not in sys.path:
@@ -759,14 +584,12 @@ class GeekyLatentSyncNode:
torch.cuda.empty_cache()
# Check and prevent ComfyUI temp creation again
comfyui_temp = get_comfyui_temp_dir()
if comfyui_temp and os.path.exists(comfyui_temp):
if os.path.exists(comfyui_temp):
try:
os.rename(comfyui_temp, f"{comfyui_temp}_backup_{uuid.uuid4().hex[:8]}")
except:
pass
log_timing("Importing inference module")
# Import the inference module
inference_module = import_inference_script(inference_script_path)
@@ -778,11 +601,9 @@ class GeekyLatentSyncNode:
inference_temp = os.path.join(temp_dir, "temp")
os.makedirs(inference_temp, exist_ok=True)
log_timing("Running Geeky inference")
# Run inference
inference_module.main(config, args)
log_timing("Processing output")
# Clean GPU cache after inference
if torch.cuda.is_available():
torch.cuda.empty_cache()
@@ -792,7 +613,6 @@ class GeekyLatentSyncNode:
raise FileNotFoundError(f"Output video not found at: {output_video_path}")
# Read the processed video - ensure it's loaded as CPU tensor
import torchvision.io as io
processed_frames = io.read_video(output_video_path, pts_unit='sec')[0]
processed_frames = processed_frames.float() / 255.0
@@ -803,55 +623,39 @@ class GeekyLatentSyncNode:
if hasattr(processed_frames, 'device') and processed_frames.device.type == 'cuda':
processed_frames = processed_frames.cpu()
total_time = time.time() - start_time
print(f"Geeky total processing time: {total_time:.2f}s")
return (processed_frames, resampled_audio)
except Exception as e:
print(f"Error during Geeky inference: {str(e)}")
print(f"Error during inference: {str(e)}")
import traceback
traceback.print_exc()
raise
finally:
# Cleanup GPU memory
if torch.cuda.is_available():
torch.cuda.empty_cache()
# Clean up the temporary symlink/copy created for inference script compatibility
try:
old_checkpoints_dir = os.path.join(os.path.dirname(os.path.abspath(__file__)), "checkpoints")
if os.path.exists(old_checkpoints_dir):
if os.path.islink(old_checkpoints_dir):
os.unlink(old_checkpoints_dir)
print("Cleaned up temporary symlink")
elif os.path.isdir(old_checkpoints_dir):
# Only remove if it's our temporary copy (check if it contains geeky files)
geeky_marker = os.path.join(old_checkpoints_dir, ".geeky_temp_copy")
if os.path.exists(geeky_marker):
shutil.rmtree(old_checkpoints_dir)
print("Cleaned up temporary copy")
except:
pass # Ignore cleanup errors
# Only remove temporary files if successful (keep for debugging if failed)
try:
# Clean up temporary files individually
for path in [temp_video_path, output_video_path, audio_path]:
if path and os.path.exists(path):
try:
os.remove(path)
except:
pass
except:
pass # Ignore cleanup errors
# Clean up temporary files individually
for path in [temp_video_path, output_video_path, audio_path]:
if path and os.path.exists(path):
try:
os.remove(path)
print(f"Removed temporary file: {path}")
except Exception as e:
print(f"Failed to remove {path}: {str(e)}")
# Remove temporary run directory
if temp_dir and os.path.exists(temp_dir):
try:
shutil.rmtree(temp_dir, ignore_errors=True)
print(f"Removed run temporary directory: {temp_dir}")
except Exception as e:
print(f"Failed to remove temp run directory: {str(e)}")
# Clean up any ComfyUI temp directories again (in case they were created during execution)
cleanup_comfyui_temp_directories()
# Final GPU cache cleanup
if torch.cuda.is_available():
torch.cuda.empty_cache()
class GeekyVideoLengthAdjuster:
@classmethod
def INPUT_TYPES(s):
@@ -954,8 +758,8 @@ NODE_CLASS_MAPPINGS = {
"GeekyVideoLengthAdjuster": GeekyVideoLengthAdjuster,
}
# Display Names for ComfyUI - Clear distinction from original
# Display Names for ComfyUI
NODE_DISPLAY_NAME_MAPPINGS = {
"GeekyLatentSyncNode": "Geeky LatentSync 1.5 (Optimized)",
"GeekyVideoLengthAdjuster": "Geeky Video Length Adjuster (Fast)",
"GeekyLatentSyncNode": "Geeky LatentSync 1.5",
"GeekyVideoLengthAdjuster": "Geeky Video Length Adjuster",
}
-85
View File
@@ -1,85 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import argparse
import os
from preprocess.affine_transform import affine_transform_multi_gpus
from preprocess.remove_broken_videos import remove_broken_videos_multiprocessing
from preprocess.detect_shot import detect_shot_multiprocessing
from preprocess.filter_high_resolution import filter_high_resolution_multiprocessing
from preprocess.resample_fps_hz import resample_fps_hz_multiprocessing
from preprocess.segment_videos import segment_videos_multiprocessing
from preprocess.sync_av import sync_av_multi_gpus
from preprocess.filter_visual_quality import filter_visual_quality_multi_gpus
from preprocess.remove_incorrect_affined import remove_incorrect_affined_multiprocessing
def data_processing_pipeline(
total_num_workers, per_gpu_num_workers, resolution, sync_conf_threshold, temp_dir, input_dir
):
print("Removing broken videos...")
remove_broken_videos_multiprocessing(input_dir, total_num_workers)
print("Resampling FPS hz...")
resampled_dir = os.path.join(os.path.dirname(input_dir), "resampled")
resample_fps_hz_multiprocessing(input_dir, resampled_dir, total_num_workers)
print("Detecting shot...")
shot_dir = os.path.join(os.path.dirname(input_dir), "shot")
detect_shot_multiprocessing(resampled_dir, shot_dir, total_num_workers)
print("Segmenting videos...")
segmented_dir = os.path.join(os.path.dirname(input_dir), "segmented")
segment_videos_multiprocessing(shot_dir, segmented_dir, total_num_workers)
# print("Filtering high resolution...")
# high_resolution_dir = os.path.join(os.path.dirname(input_dir), "high_resolution")
# filter_high_resolution_multiprocessing(segmented_dir, high_resolution_dir, resolution, total_num_workers)
print("Affine transforming videos...")
affine_transformed_dir = os.path.join(os.path.dirname(input_dir), "affine_transformed")
affine_transform_multi_gpus(
segmented_dir, affine_transformed_dir, temp_dir, resolution, per_gpu_num_workers // 2
)
print("Removing incorrect affined videos...")
remove_incorrect_affined_multiprocessing(affine_transformed_dir, total_num_workers)
print("Syncing audio and video...")
av_synced_dir = os.path.join(os.path.dirname(input_dir), f"av_synced_{sync_conf_threshold}")
sync_av_multi_gpus(affine_transformed_dir, av_synced_dir, temp_dir, per_gpu_num_workers, sync_conf_threshold)
print("Filtering visual quality...")
high_visual_quality_dir = os.path.join(os.path.dirname(input_dir), "high_visual_quality")
filter_visual_quality_multi_gpus(av_synced_dir, high_visual_quality_dir, per_gpu_num_workers)
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument("--total_num_workers", type=int, default=100)
parser.add_argument("--per_gpu_num_workers", type=int, default=20)
parser.add_argument("--resolution", type=int, default=256)
parser.add_argument("--sync_conf_threshold", type=int, default=3)
parser.add_argument("--temp_dir", type=str, default="temp")
parser.add_argument("--input_dir", type=str, required=True)
args = parser.parse_args()
data_processing_pipeline(
args.total_num_workers,
args.per_gpu_num_workers,
args.resolution,
args.sync_conf_threshold,
args.temp_dir,
args.input_dir,
)
-62
View File
@@ -1,62 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
import subprocess
import tqdm
from multiprocessing import Pool
paths = []
def gather_paths(input_dir, output_dir):
for video in sorted(os.listdir(input_dir)):
if video.endswith(".mp4"):
video_input = os.path.join(input_dir, video)
video_output = os.path.join(output_dir, video)
if os.path.isfile(video_output):
continue
paths.append([video_input, output_dir])
elif os.path.isdir(os.path.join(input_dir, video)):
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
def detect_shot(video_input, output_dir):
os.makedirs(output_dir, exist_ok=True)
video = os.path.basename(video_input)[:-4]
command = f"scenedetect --quiet -i {video_input} detect-adaptive --threshold 2 split-video --filename '{video}_shot_$SCENE_NUMBER' --output {output_dir}"
# command = f"scenedetect --quiet -i {video_input} detect-adaptive --threshold 2 split-video --high-quality --filename '{video}_shot_$SCENE_NUMBER' --output {output_dir}"
subprocess.run(command, shell=True)
def multi_run_wrapper(args):
return detect_shot(*args)
def detect_shot_multiprocessing(input_dir, output_dir, num_workers):
print(f"Recursively gathering video paths of {input_dir} ...")
gather_paths(input_dir, output_dir)
print(f"Detecting shot of {input_dir} ...")
with Pool(num_workers) as pool:
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
pass
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/ads/high-resolution"
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/ads/shot"
num_workers = 50
detect_shot_multiprocessing(input_dir, output_dir, num_workers)
-112
View File
@@ -1,112 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import mediapipe as mp
from latentsync.utils.util import read_video
import os
import tqdm
import shutil
from multiprocessing import Pool
paths = []
def gather_video_paths(input_dir, output_dir, resolution):
for video in sorted(os.listdir(input_dir)):
if video.endswith(".mp4"):
video_input = os.path.join(input_dir, video)
video_output = os.path.join(output_dir, video)
if os.path.isfile(video_output):
continue
paths.append([video_input, video_output, resolution])
elif os.path.isdir(os.path.join(input_dir, video)):
gather_video_paths(os.path.join(input_dir, video), os.path.join(output_dir, video), resolution)
class FaceDetector:
def __init__(self, resolution=256):
self.face_detection = mp.solutions.face_detection.FaceDetection(
model_selection=0, min_detection_confidence=0.5
)
self.resolution = resolution
def detect_face(self, image):
height, width = image.shape[:2]
# Process the image and detect faces.
results = self.face_detection.process(image)
if not results.detections: # Face not detected
raise Exception("Face not detected")
if len(results.detections) != 1:
return False
detection = results.detections[0] # Only use the first face in the image
bounding_box = detection.location_data.relative_bounding_box
face_width = int(bounding_box.width * width)
face_height = int(bounding_box.height * height)
if face_width < self.resolution or face_height < self.resolution:
return False
return True
def detect_video(self, video_path):
video_frames = read_video(video_path, change_fps=False)
if len(video_frames) == 0:
return False
for frame in video_frames:
if not self.detect_face(frame):
return False
return True
def close(self):
self.face_detection.close()
def filter_video(video_input, video_out, resolution):
if os.path.isfile(video_out):
return
face_detector = FaceDetector(resolution)
try:
save = face_detector.detect_video(video_input)
except Exception as e:
# print(f"Exception: {e} Input video: {video_input}")
face_detector.close()
return
if save:
os.makedirs(os.path.dirname(video_out), exist_ok=True)
shutil.copy(video_input, video_out)
face_detector.close()
def multi_run_wrapper(args):
return filter_video(*args)
def filter_high_resolution_multiprocessing(input_dir, output_dir, resolution, num_workers):
print(f"Recursively gathering video paths of {input_dir} ...")
gather_video_paths(input_dir, output_dir, resolution)
print(f"Filtering high resolution videos in {input_dir} ...")
with Pool(num_workers) as pool:
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
pass
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai/lichunyu/HDTF/original/train"
output_dir = "/mnt/bn/maliva-gen-ai/lichunyu/HDTF/detected/train"
resolution = 256
num_workers = 50
filter_high_resolution_multiprocessing(input_dir, output_dir, resolution, num_workers)
-129
View File
@@ -1,129 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
import tqdm
import torch
import torchvision
import shutil
from multiprocessing import Process
import numpy as np
from decord import VideoReader
from einops import rearrange
from eval.hyper_iqa import HyperNet, TargetNet
paths = []
def gather_paths(input_dir, output_dir):
# os.makedirs(output_dir, exist_ok=True)
for video in tqdm.tqdm(sorted(os.listdir(input_dir))):
if video.endswith(".mp4"):
video_input = os.path.join(input_dir, video)
video_output = os.path.join(output_dir, video)
if os.path.isfile(video_output):
continue
paths.append((video_input, video_output))
elif os.path.isdir(os.path.join(input_dir, video)):
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
def read_video(video_path: str):
vr = VideoReader(video_path)
first_frame = vr[0].asnumpy()
middle_frame = vr[len(vr) // 2].asnumpy()
last_frame = vr[-1].asnumpy()
vr.seek(0)
video_frames = np.stack([first_frame, middle_frame, last_frame], axis=0)
video_frames = torch.from_numpy(rearrange(video_frames, "b h w c -> b c h w"))
video_frames = video_frames / 255.0
return video_frames
def func(paths, device_id):
device = f"cuda:{device_id}"
model_hyper = HyperNet(16, 112, 224, 112, 56, 28, 14, 7).to(device)
model_hyper.train(False)
# load the pre-trained model on the koniq-10k dataset
model_hyper.load_state_dict(
(torch.load("checkpoints/auxiliary/koniq_pretrained.pkl", map_location=device, weights_only=True))
)
transforms = torchvision.transforms.Compose(
[
torchvision.transforms.CenterCrop(size=224),
torchvision.transforms.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),
]
)
for video_input, video_output in paths:
try:
video_frames = read_video(video_input)
video_frames = transforms(video_frames)
video_frames = video_frames.clone().detach().to(device)
paras = model_hyper(video_frames) # 'paras' contains the network weights conveyed to target network
# Building target network
model_target = TargetNet(paras).to(device)
for param in model_target.parameters():
param.requires_grad = False
# Quality prediction
pred = model_target(paras["target_in_vec"]) # 'paras['target_in_vec']' is the input to target net
# quality score ranges from 0-100, a higher score indicates a better quality
quality_score = pred.mean().item()
print(f"Input video: {video_input}\nVisual quality score: {quality_score:.2f}")
if quality_score >= 40:
os.makedirs(os.path.dirname(video_output), exist_ok=True)
shutil.copy(video_input, video_output)
except Exception as e:
print(e)
def split(a, n):
k, m = divmod(len(a), n)
return (a[i * k + min(i, m) : (i + 1) * k + min(i + 1, m)] for i in range(n))
def filter_visual_quality_multi_gpus(input_dir, output_dir, num_workers):
gather_paths(input_dir, output_dir)
num_devices = torch.cuda.device_count()
if num_devices == 0:
raise RuntimeError("No GPUs found")
split_paths = list(split(paths, num_workers * num_devices))
processes = []
for i in range(num_devices):
for j in range(num_workers):
process_index = i * num_workers + j
process = Process(target=func, args=(split_paths[process_index], i))
process.start()
processes.append(process)
for process in processes:
process.join()
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/VoxCeleb2/av_synced_high"
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/VoxCeleb2/high_visual_quality"
num_workers = 20 # How many processes per device
filter_visual_quality_multi_gpus(input_dir, output_dir, num_workers)
-43
View File
@@ -1,43 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
from multiprocessing import Pool
import tqdm
from latentsync.utils.av_reader import AVReader
from latentsync.utils.util import gather_video_paths_recursively
def remove_broken_video(video_path):
try:
AVReader(video_path)
except Exception:
os.remove(video_path)
def remove_broken_videos_multiprocessing(input_dir, num_workers):
video_paths = gather_video_paths_recursively(input_dir)
print("Removing broken videos...")
with Pool(num_workers) as pool:
for _ in tqdm.tqdm(pool.imap_unordered(remove_broken_video, video_paths), total=len(video_paths)):
pass
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual/affine_transformed"
num_workers = 50
remove_broken_videos_multiprocessing(input_dir, num_workers)
-81
View File
@@ -1,81 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import mediapipe as mp
from latentsync.utils.util import read_video, gather_video_paths_recursively
import os
import tqdm
from multiprocessing import Pool
class FaceDetector:
def __init__(self):
self.face_detection = mp.solutions.face_detection.FaceDetection(
model_selection=0, min_detection_confidence=0.5
)
def detect_face(self, image):
# Process the image and detect faces.
results = self.face_detection.process(image)
if not results.detections: # Face not detected
return False
if len(results.detections) != 1:
return False
return True
def detect_video(self, video_path):
try:
video_frames = read_video(video_path, change_fps=False)
except Exception as e:
print(f"Exception: {e} - {video_path}")
return False
if len(video_frames) == 0:
return False
for frame in video_frames:
if not self.detect_face(frame):
return False
return True
def close(self):
self.face_detection.close()
def remove_incorrect_affined(video_path):
if not os.path.isfile(video_path):
return
face_detector = FaceDetector()
has_face = face_detector.detect_video(video_path)
if not has_face:
os.remove(video_path)
print(f"Removed: {video_path}")
face_detector.close()
def remove_incorrect_affined_multiprocessing(input_dir, num_workers):
video_paths = gather_video_paths_recursively(input_dir)
print(f"Total videos: {len(video_paths)}")
print(f"Removing incorrect affined videos in {input_dir} ...")
with Pool(num_workers) as pool:
for _ in tqdm.tqdm(pool.imap_unordered(remove_incorrect_affined, video_paths), total=len(video_paths)):
pass
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual_dcc/high_visual_quality"
num_workers = 50
remove_incorrect_affined_multiprocessing(input_dir, num_workers)
-70
View File
@@ -1,70 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
import subprocess
import tqdm
from multiprocessing import Pool
import cv2
paths = []
def gather_paths(input_dir, output_dir):
for video in sorted(os.listdir(input_dir)):
if video.endswith(".mp4"):
video_input = os.path.join(input_dir, video)
video_output = os.path.join(output_dir, video)
if os.path.isfile(video_output):
continue
paths.append([video_input, video_output])
elif os.path.isdir(os.path.join(input_dir, video)):
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
def get_video_fps(video_path: str):
cam = cv2.VideoCapture(video_path)
fps = cam.get(cv2.CAP_PROP_FPS)
return fps
def resample_fps_hz(video_input, video_output):
os.makedirs(os.path.dirname(video_output), exist_ok=True)
if get_video_fps(video_input) == 25:
command = f"ffmpeg -loglevel error -y -i {video_input} -c:v copy -ar 16000 -q:a 0 {video_output}"
else:
command = f"ffmpeg -loglevel error -y -i {video_input} -r 25 -ar 16000 -q:a 0 {video_output}"
subprocess.run(command, shell=True)
def multi_run_wrapper(args):
return resample_fps_hz(*args)
def resample_fps_hz_multiprocessing(input_dir, output_dir, num_workers):
print(f"Recursively gathering video paths of {input_dir} ...")
gather_paths(input_dir, output_dir)
print(f"Resampling FPS and Hz of {input_dir} ...")
with Pool(num_workers) as pool:
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
pass
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/HDTF/segmented/train"
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/HDTF/resampled_test"
num_workers = 20
resample_fps_hz_multiprocessing(input_dir, output_dir, num_workers)
-62
View File
@@ -1,62 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
import subprocess
import tqdm
from multiprocessing import Pool
paths = []
def gather_paths(input_dir, output_dir):
for video in sorted(os.listdir(input_dir)):
if video.endswith(".mp4"):
video_basename = video[:-4]
video_input = os.path.join(input_dir, video)
video_output = os.path.join(output_dir, f"{video_basename}_%03d.mp4")
if os.path.isfile(video_output):
continue
paths.append([video_input, video_output])
elif os.path.isdir(os.path.join(input_dir, video)):
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
def segment_video(video_input, video_output):
os.makedirs(os.path.dirname(video_output), exist_ok=True)
command = f"ffmpeg -loglevel error -y -i {video_input} -map 0 -c:v copy -segment_time 5 -f segment -reset_timestamps 1 -q:a 0 {video_output}"
# command = f'ffmpeg -loglevel error -y -i {video_input} -map 0 -segment_time 5 -f segment -reset_timestamps 1 -force_key_frames "expr:gte(t,n_forced*5)" -crf 18 -q:a 0 {video_output}'
subprocess.run(command, shell=True)
def multi_run_wrapper(args):
return segment_video(*args)
def segment_videos_multiprocessing(input_dir, output_dir, num_workers):
print(f"Recursively gathering video paths of {input_dir} ...")
gather_paths(input_dir, output_dir)
print(f"Segmenting videos of {input_dir} ...")
with Pool(num_workers) as pool:
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
pass
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/avatars_new/cut"
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/avatars_new/segmented"
num_workers = 50
segment_videos_multiprocessing(input_dir, output_dir, num_workers)
-114
View File
@@ -1,114 +0,0 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
import tqdm
from eval.syncnet import SyncNetEval
from eval.syncnet_detect import SyncNetDetector
from eval.eval_sync_conf import syncnet_eval
import torch
import subprocess
import shutil
from multiprocessing import Process
paths = []
def gather_paths(input_dir, output_dir):
# os.makedirs(output_dir, exist_ok=True)
for video in tqdm.tqdm(sorted(os.listdir(input_dir))):
if video.endswith(".mp4"):
video_input = os.path.join(input_dir, video)
video_output = os.path.join(output_dir, video)
if os.path.isfile(video_output):
continue
paths.append((video_input, video_output))
elif os.path.isdir(os.path.join(input_dir, video)):
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
def adjust_offset(video_input: str, video_output: str, av_offset: int, fps: int = 25):
command = f"ffmpeg -loglevel error -y -i {video_input} -itsoffset {av_offset/fps} -i {video_input} -map 0:v -map 1:a -c copy -q:v 0 -q:a 0 {video_output}"
subprocess.run(command, shell=True)
def func(sync_conf_threshold, paths, device_id, process_temp_dir):
os.makedirs(process_temp_dir, exist_ok=True)
device = f"cuda:{device_id}"
syncnet = SyncNetEval(device=device)
syncnet.loadParameters("checkpoints/auxiliary/syncnet_v2.model")
detect_results_dir = os.path.join(process_temp_dir, "detect_results")
syncnet_eval_results_dir = os.path.join(process_temp_dir, "syncnet_eval_results")
syncnet_detector = SyncNetDetector(device=device, detect_results_dir=detect_results_dir)
for video_input, video_output in paths:
try:
av_offset, conf = syncnet_eval(
syncnet, syncnet_detector, video_input, syncnet_eval_results_dir, detect_results_dir
)
if conf >= sync_conf_threshold and abs(av_offset) <= 6:
os.makedirs(os.path.dirname(video_output), exist_ok=True)
if av_offset == 0:
shutil.copy(video_input, video_output)
else:
adjust_offset(video_input, video_output, av_offset)
except Exception as e:
print(e)
def split(a, n):
k, m = divmod(len(a), n)
return (a[i * k + min(i, m) : (i + 1) * k + min(i + 1, m)] for i in range(n))
def sync_av_multi_gpus(input_dir, output_dir, temp_dir, num_workers, sync_conf_threshold):
gather_paths(input_dir, output_dir)
num_devices = torch.cuda.device_count()
if num_devices == 0:
raise RuntimeError("No GPUs found")
split_paths = list(split(paths, num_workers * num_devices))
processes = []
for i in range(num_devices):
for j in range(num_workers):
process_index = i * num_workers + j
process = Process(
target=func,
args=(
sync_conf_threshold,
split_paths[process_index],
i,
os.path.join(temp_dir, f"process_{process_index}"),
),
)
process.start()
processes.append(process)
for process in processes:
process.join()
if __name__ == "__main__":
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/ads/affine_transformed"
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/VoxCeleb2/temp"
temp_dir = "temp"
num_workers = 20 # How many processes per device
sync_conf_threshold = 3
sync_av_multi_gpus(input_dir, output_dir, temp_dir, num_workers, sync_conf_threshold)
+113 -113
View File
@@ -1,113 +1,113 @@
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
import subprocess
from concurrent.futures import ThreadPoolExecutor
import pandas as pd
from tqdm import tqdm
"""
To use this python file, first install yt-dlp by:
pip install yt-dlp==2024.5.27
"""
def download_video(video_url, video_path):
get_video_channel_command = f"yt-dlp --print channel {video_url}"
result = subprocess.run(get_video_channel_command, shell=True, capture_output=True, text=True)
channel = result.stdout.strip()
if channel in unwanted_channels:
return
download_video_command = f"yt-dlp -f bestvideo+bestaudio --skip-unavailable-fragments --merge-output-format mp4 '{video_url}' --output '{video_path}' --external-downloader aria2c --external-downloader-args '-x 16 -k 1M'"
try:
subprocess.run(download_video_command, shell=True) # ignore_security_alert_wait_for_fix RCE
except KeyboardInterrupt:
print("Stopped")
exit()
except:
print(f"Error downloading video {video_url}")
def download_videos(num_workers, video_urls, video_paths):
with ThreadPoolExecutor(max_workers=num_workers) as executor:
executor.map(download_video, video_urls, video_paths)
def read_video_urls(csv_file_path: str, language_column, video_url_column):
video_urls = []
print("Reading video urls...")
df = pd.read_csv(csv_file_path, sep=",")
for row in tqdm(df.itertuples(), total=len(df)):
language = getattr(row, language_column)
video_url = getattr(row, video_url_column)
if "clip" in video_url:
continue
video_urls.append((language, video_url))
return video_urls
def extract_vid(video_url):
if "watch?v=" in video_url: # ignore_security_alert_wait_for_fix RCE
return video_url.split("watch?v=")[1][:11]
elif "shorts/" in video_url:
return video_url.split("shorts/")[1][:11]
elif "youtu.be/" in video_url:
return video_url.split("youtu.be/")[1][:11]
elif "&v=" in video_url:
return video_url.split("&v=")[1][:11]
else:
print(f"Invalid video url: {video_url}")
return None
def main(csv_file_path, language_column, video_url_column, output_dir, num_workers):
os.makedirs(output_dir, exist_ok=True)
all_video_urls = read_video_urls(csv_file_path, language_column, video_url_column)
video_paths = []
video_urls = []
print("Extracting vid...")
for language, video_url in tqdm(all_video_urls):
vid = extract_vid(video_url)
if vid is None:
continue
video_path = os.path.join(output_dir, language.lower(), f"vid_{vid}.mp4")
if os.path.isfile(video_path):
continue
os.makedirs(os.path.dirname(video_path), exist_ok=True)
video_paths.append(video_path)
video_urls.append(video_url)
if len(video_paths) == 0:
print("All videos have been downloaded")
exit()
else:
print(f"Downloading {len(video_paths)} videos")
download_videos(num_workers, video_urls, video_paths)
if __name__ == "__main__":
csv_file_path = "dcc.csv"
language_column = "video_language"
video_url_column = "video_link"
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual/raw"
num_workers = 50
unwanted_channels = ["TEDx Talks", "DaePyeong Mukbang", "Joeman"]
main(csv_file_path, language_column, video_url_column, output_dir, num_workers)
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
import subprocess
from concurrent.futures import ThreadPoolExecutor
import pandas as pd
from tqdm import tqdm
"""
To use this python file, first install yt-dlp by:
pip install yt-dlp==2024.5.27
"""
def download_video(video_url, video_path):
get_video_channel_command = f"yt-dlp --print channel {video_url}"
result = subprocess.run(get_video_channel_command, shell=True, capture_output=True, text=True)
channel = result.stdout.strip()
if channel in unwanted_channels:
return
download_video_command = f"yt-dlp -f bestvideo+bestaudio --skip-unavailable-fragments --merge-output-format mp4 '{video_url}' --output '{video_path}' --external-downloader aria2c --external-downloader-args '-x 16 -k 1M'"
try:
subprocess.run(download_video_command, shell=True) # ignore_security_alert_wait_for_fix RCE
except KeyboardInterrupt:
print("Stopped")
exit()
except:
print(f"Error downloading video {video_url}")
def download_videos(num_workers, video_urls, video_paths):
with ThreadPoolExecutor(max_workers=num_workers) as executor:
executor.map(download_video, video_urls, video_paths)
def read_video_urls(csv_file_path: str, language_column, video_url_column):
video_urls = []
print("Reading video urls...")
df = pd.read_csv(csv_file_path, sep=",")
for row in tqdm(df.itertuples(), total=len(df)):
language = getattr(row, language_column)
video_url = getattr(row, video_url_column)
if "clip" in video_url:
continue
video_urls.append((language, video_url))
return video_urls
def extract_vid(video_url):
if "watch?v=" in video_url: # ignore_security_alert_wait_for_fix RCE
return video_url.split("watch?v=")[1][:11]
elif "shorts/" in video_url:
return video_url.split("shorts/")[1][:11]
elif "youtu.be/" in video_url:
return video_url.split("youtu.be/")[1][:11]
elif "&v=" in video_url:
return video_url.split("&v=")[1][:11]
else:
print(f"Invalid video url: {video_url}")
return None
def main(csv_file_path, language_column, video_url_column, output_dir, num_workers):
os.makedirs(output_dir, exist_ok=True)
all_video_urls = read_video_urls(csv_file_path, language_column, video_url_column)
video_paths = []
video_urls = []
print("Extracting vid...")
for language, video_url in tqdm(all_video_urls):
vid = extract_vid(video_url)
if vid is None:
continue
video_path = os.path.join(output_dir, language.lower(), f"vid_{vid}.mp4")
if os.path.isfile(video_path):
continue
os.makedirs(os.path.dirname(video_path), exist_ok=True)
video_paths.append(video_path)
video_urls.append(video_url)
if len(video_paths) == 0:
print("All videos have been downloaded")
exit()
else:
print(f"Downloading {len(video_paths)} videos")
download_videos(num_workers, video_urls, video_paths)
if __name__ == "__main__":
csv_file_path = "dcc.csv"
language_column = "video_language"
video_url_column = "video_link"
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual/raw"
num_workers = 50
unwanted_channels = ["TEDx Talks", "DaePyeong Mukbang", "Joeman"]
main(csv_file_path, language_column, video_url_column, output_dir, num_workers)
+273
View File
@@ -0,0 +1,273 @@
{
"last_node_id": 5,
"last_link_id": 5,
"nodes": [
{
"id": 4,
"type": "VHS_VideoCombine",
"pos": [
-877.4164428710938,
130.14813232421875
],
"size": [
290.8520202636719,
709.1360473632812
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 3
},
{
"name": "audio",
"type": "AUDIO",
"shape": 7,
"link": 4
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"shape": 7,
"link": null
},
{
"name": "vae",
"type": "VAE",
"shape": 7,
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "1.5.9",
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 25,
"loop_count": 0,
"filename_prefix": "LatentSync",
"format": "video/nvenc_h264-mp4",
"pix_fmt": "yuv420p",
"bitrate": 10,
"megabit": true,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "LatentSync_00004-audio.mp4",
"subfolder": "",
"type": "output",
"format": "video/nvenc_h264-mp4",
"frame_rate": 25,
"workflow": "LatentSync_00004.png",
"fullpath": "C:\\Users\\wgray\\Documents\\ComfyUI test branch\\ComfyUI_windows_portable\\ComfyUI\\output\\LatentSync_00004-audio.mp4"
}
}
}
},
{
"id": 1,
"type": "GeekyKokoroTTS",
"pos": [
-1664.339599609375,
132.1678924560547
],
"size": [
400,
252
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "audio",
"type": "AUDIO",
"links": [
2
],
"slot_index": 0
},
{
"name": "text_processed",
"type": "STRING",
"links": null
}
],
"properties": {
"cnr_id": "ComfyUI-Geeky-Kokoro-TTS",
"ver": "443f4b0e1c503b47737b18c89f5067b9a6187038",
"Node name for S&R": "GeekyKokoroTTS"
},
"widgets_values": [
"Welcome to ComfyUI",
"🇺🇸 🚺 Heart ❤️",
1,
true,
false,
"🇺🇸 🚺 Sarah",
0.5
]
},
{
"id": 3,
"type": "LatentSyncNode",
"pos": [
-1224.6712646484375,
134.58547973632812
],
"size": [
315,
150
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 5
},
{
"name": "audio",
"type": "AUDIO",
"link": 2
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
3
],
"slot_index": 0
},
{
"name": "audio",
"type": "AUDIO",
"links": [
4
],
"slot_index": 1
}
],
"properties": {
"aux_id": "GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper",
"ver": "d4c98e2a3718dba91fc87098640db6fea8d86409",
"Node name for S&R": "LatentSyncNode"
},
"widgets_values": [
743,
"randomize",
1.5,
20
]
},
{
"id": 5,
"type": "LoadImage",
"pos": [
-2015.361083984375,
147.8274688720703
],
"size": [
315,
314
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
5
]
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"cnr_id": "comfy-core",
"ver": "0.3.26",
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"ComfyUI_00611_.png",
"image"
]
}
],
"links": [
[
2,
1,
0,
3,
1,
"AUDIO"
],
[
3,
3,
0,
4,
0,
"IMAGE"
],
[
4,
3,
1,
4,
1,
"AUDIO"
],
[
5,
5,
0,
3,
0,
"IMAGE"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.9646149645000188,
"offset": [
2123.341499155334,
21.38434610981646
]
},
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
+311 -349
View File
@@ -1,349 +1,311 @@
{
"id": "0bed89c3-ac44-45f4-b651-66c7cec8527b",
"revision": 0,
"last_node_id": 5,
"last_link_id": 4,
"nodes": [
{
"id": 1,
"type": "GeekyLatentSyncNode",
"pos": [
4711.150390625,
1535.1771240234375
],
"size": [
313.8853454589844,
174
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 1
},
{
"name": "audio",
"type": "AUDIO",
"link": 2
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
3
]
},
{
"name": "audio",
"type": "AUDIO",
"links": [
4
]
}
],
"properties": {
"cnr_id": "ComfyUI-LatentSyncWrapper",
"ver": "c467ec665a7bfad3fa3052134fbddb33f8fc58ad",
"Node name for S&R": "GeekyLatentSyncNode"
},
"widgets_values": [
312,
"randomize",
1.8,
20,
"medium"
]
},
{
"id": 3,
"type": "VHS_LoadAudioUpload",
"pos": [
4716.10595703125,
1338.88330078125
],
"size": [
243.818359375,
130
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "audio",
"type": "AUDIO",
"links": [
2
]
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "1.5.9",
"Node name for S&R": "VHS_LoadAudioUpload"
},
"widgets_values": {
"audio": "ComfyUI_temp_efoos_00005_.flac",
"start_time": 0,
"duration": 0,
"choose audio to upload": "image"
}
},
{
"id": 4,
"type": "VHS_LoadVideo",
"pos": [
4438.70849609375,
1317.5367431640625
],
"size": [
247.455078125,
494.59130859375
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"shape": 7,
"type": "VHS_BatchManager",
"link": null
},
{
"name": "vae",
"shape": 7,
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1
]
},
{
"name": "frame_count",
"type": "INT",
"links": null
},
{
"name": "audio",
"type": "AUDIO",
"links": null
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": []
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "1.5.9",
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "Upscaled_Wan_00004.mp4",
"force_rate": 25,
"custom_width": 0,
"custom_height": 0,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1,
"format": "AnimateDiff",
"choose video to upload": "image",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "Upscaled_Wan_00004.mp4",
"type": "input",
"format": "video/mp4",
"force_rate": 25,
"custom_width": 0,
"custom_height": 0,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1
}
}
}
},
{
"id": 5,
"type": "VHS_VideoCombine",
"pos": [
5047.68896484375,
1320.9598388671875
],
"size": [
450.99688720703125,
671.2476806640625
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 3
},
{
"name": "audio",
"shape": 7,
"type": "AUDIO",
"link": 4
},
{
"name": "meta_batch",
"shape": 7,
"type": "VHS_BatchManager",
"link": null
},
{
"name": "vae",
"shape": 7,
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "1.5.9",
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 25,
"loop_count": 0,
"filename_prefix": "LatentSync",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": false,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "LatentSync_00003-audio.mp4",
"subfolder": "",
"type": "temp",
"format": "video/h264-mp4",
"frame_rate": 25,
"workflow": "LatentSync_00003.png",
"fullpath": "C:\\Users\\wgray\\AppData\\Local\\Temp\\geeky_latentsync_e108de4d\\latentsync_43dc5e26\\latentsync_76385145\\geeky_latentsync_5bfbc51d\\latentsync_7c3d9782\\latentsync_a2ca1c39\\LatentSync_00003-audio.mp4"
}
}
}
},
{
"id": 2,
"type": "Note",
"pos": [
4436.41162109375,
1878.3404541015625
],
"size": [
395.8923034667969,
100.62284851074219
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"LatentSync can be found with comfyUI manager or here for a faster 1.5 version\n\nhttps://github.com/GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper.git"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
1,
4,
0,
1,
0,
"IMAGE"
],
[
2,
3,
0,
1,
1,
"AUDIO"
],
[
3,
1,
0,
5,
0,
"IMAGE"
],
[
4,
1,
1,
5,
1,
"AUDIO"
]
],
"groups": [
{
"id": 1,
"title": "Lip Sync Pass",
"bounding": [
4405.30322265625,
1221.0234375,
1110.41650390625,
783.9459228515625
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ue_links": [],
"ds": {
"scale": 0.8769226950000164,
"offset": [
-4066.528439416358,
-1092.0980920410036
]
},
"frontendVersion": "1.23.4",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
{
"last_node_id": 4,
"last_link_id": 4,
"nodes": [
{
"id": 3,
"type": "LatentSyncNode",
"pos": [
-1224.6712646484375,
134.58547973632812
],
"size": [
315,
150
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 1
},
{
"name": "audio",
"type": "AUDIO",
"link": 2
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
3
],
"slot_index": 0
},
{
"name": "audio",
"type": "AUDIO",
"links": [
4
],
"slot_index": 1
}
],
"properties": {
"aux_id": "GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper",
"ver": "d4c98e2a3718dba91fc87098640db6fea8d86409",
"Node name for S&R": "LatentSyncNode"
},
"widgets_values": [
991,
"randomize",
1.5,
20
]
},
{
"id": 4,
"type": "VHS_VideoCombine",
"pos": [
-877.4164428710938,
130.14813232421875
],
"size": [
290.8520202636719,
334
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 3
},
{
"name": "audio",
"type": "AUDIO",
"shape": 7,
"link": 4
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"shape": 7,
"link": null
},
{
"name": "vae",
"type": "VAE",
"shape": 7,
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "1.5.9",
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 25,
"loop_count": 0,
"filename_prefix": "LatentSync",
"format": "video/nvenc_h264-mp4",
"pix_fmt": "yuv420p",
"bitrate": 10,
"megabit": true,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {}
}
}
},
{
"id": 2,
"type": "VHS_LoadVideo",
"pos": [
-1940.4915771484375,
144.70785522460938
],
"size": [
247.455078125,
627.8657836914062
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"shape": 7,
"link": null
},
{
"name": "vae",
"type": "VAE",
"shape": 7,
"link": null
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1
],
"slot_index": 0
},
{
"name": "frame_count",
"type": "INT",
"links": null
},
{
"name": "audio",
"type": "AUDIO",
"links": null
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": null
}
],
"properties": {
"cnr_id": "comfyui-videohelpersuite",
"ver": "1.5.9",
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "AnimateDiff_00381.mp4",
"force_rate": 0,
"custom_width": 0,
"custom_height": 0,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1,
"format": "AnimateDiff",
"choose video to upload": "image",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "AnimateDiff_00381.mp4",
"type": "input",
"format": "video/mp4",
"force_rate": 0,
"custom_width": 0,
"custom_height": 0,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1
}
}
}
},
{
"id": 1,
"type": "GeekyKokoroTTS",
"pos": [
-1664.339599609375,
132.1678924560547
],
"size": [
400,
252
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "audio",
"type": "AUDIO",
"links": [
2
],
"slot_index": 0
},
{
"name": "text_processed",
"type": "STRING",
"links": null
}
],
"properties": {
"cnr_id": "ComfyUI-Geeky-Kokoro-TTS",
"ver": "443f4b0e1c503b47737b18c89f5067b9a6187038",
"Node name for S&R": "GeekyKokoroTTS"
},
"widgets_values": [
"Welcome to ComfyUI",
"🇺🇸 🚺 Heart ❤️",
1,
true,
false,
"🇺🇸 🚺 Sarah",
0.5
]
}
],
"links": [
[
1,
2,
0,
3,
0,
"IMAGE"
],
[
2,
1,
0,
3,
1,
"AUDIO"
],
[
3,
3,
0,
4,
0,
"IMAGE"
],
[
4,
3,
1,
4,
1,
"AUDIO"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.9646149645000188,
"offset": [
2147.454892156228,
60.38347479123573
]
},
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}