Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d57a11d9a8 | ||
|
|
c00bbab912 | ||
|
|
8f2a4d5e82 | ||
|
|
866c4acdde |
@@ -1,38 +0,0 @@
|
||||
---
|
||||
name: Bug Report
|
||||
about: Create a report to help us improve
|
||||
title: ''
|
||||
labels: ''
|
||||
assignees: ''
|
||||
---
|
||||
|
||||
<!--
|
||||
🔍 STOP! Before creating a new issue:
|
||||
1. Please search the closed issues first: https://github.com/ShmuelRonen/ComfyUI-LatentSyncWrapper/issues?q=is%3Aissue+is%3Aclosed
|
||||
2. Check if your issue has already been resolved in past discussions.
|
||||
3. If you find a similar issue, please add a comment there instead of creating a new one.
|
||||
-->
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**Steps To Reproduce**
|
||||
1. Go to '...'
|
||||
2. Click on '....'
|
||||
3. Scroll down to '....'
|
||||
4. See error
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Additional context**
|
||||
Add any other context about the problem here.
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
## Additional Context
|
||||
Add any other context about the problem here (screenshots, logs, etc.).
|
||||
|
||||
## Environment
|
||||
- OS: [e.g. Windows 10, Ubuntu 20.04]
|
||||
- Browser: [e.g. Chrome, Firefox]
|
||||
- Version: [e.g. 22]
|
||||
@@ -1,26 +0,0 @@
|
||||
name: Publish to Comfy registry
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
- master
|
||||
paths:
|
||||
- "pyproject.toml"
|
||||
|
||||
jobs:
|
||||
publish-node:
|
||||
name: Publish Custom Node to registry
|
||||
runs-on: ubuntu-latest
|
||||
# if this is a forked repository. Skipping the workflow.
|
||||
if: github.event.repository.fork == false
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
submodules: true
|
||||
- name: Publish Custom Node
|
||||
uses: Comfy-Org/publish-node-action@main
|
||||
with:
|
||||
## Add your own personal access token to your Github Repository secrets and reference it here.
|
||||
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
|
||||
-171
@@ -1,171 +0,0 @@
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*$py.class
|
||||
|
||||
# C extensions
|
||||
*.so
|
||||
|
||||
# Distribution / packaging
|
||||
.Python
|
||||
build/
|
||||
develop-eggs/
|
||||
dist/
|
||||
downloads/
|
||||
eggs/
|
||||
.eggs/
|
||||
lib/
|
||||
lib64/
|
||||
parts/
|
||||
sdist/
|
||||
var/
|
||||
wheels/
|
||||
share/python-wheels/
|
||||
*.egg-info/
|
||||
.installed.cfg
|
||||
*.egg
|
||||
MANIFEST
|
||||
|
||||
# PyInstaller
|
||||
# Usually these files are written by a python script from a template
|
||||
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||
*.manifest
|
||||
*.spec
|
||||
|
||||
# Installer logs
|
||||
pip-log.txt
|
||||
pip-delete-this-directory.txt
|
||||
|
||||
# Unit test / coverage reports
|
||||
htmlcov/
|
||||
.tox/
|
||||
.nox/
|
||||
.coverage
|
||||
.coverage.*
|
||||
.cache
|
||||
nosetests.xml
|
||||
coverage.xml
|
||||
*.cover
|
||||
*.py,cover
|
||||
.hypothesis/
|
||||
.pytest_cache/
|
||||
cover/
|
||||
|
||||
# Translations
|
||||
*.mo
|
||||
*.pot
|
||||
|
||||
# Django stuff:
|
||||
*.log
|
||||
local_settings.py
|
||||
db.sqlite3
|
||||
db.sqlite3-journal
|
||||
|
||||
# Flask stuff:
|
||||
instance/
|
||||
.webassets-cache
|
||||
|
||||
# Scrapy stuff:
|
||||
.scrapy
|
||||
|
||||
# Sphinx documentation
|
||||
docs/_build/
|
||||
|
||||
# PyBuilder
|
||||
.pybuilder/
|
||||
target/
|
||||
|
||||
# Jupyter Notebook
|
||||
.ipynb_checkpoints
|
||||
|
||||
# IPython
|
||||
profile_default/
|
||||
ipython_config.py
|
||||
|
||||
# pyenv
|
||||
# For a library or package, you might want to ignore these files since the code is
|
||||
# intended to run in multiple environments; otherwise, check them in:
|
||||
# .python-version
|
||||
|
||||
# pipenv
|
||||
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
||||
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
||||
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
||||
# install all needed dependencies.
|
||||
#Pipfile.lock
|
||||
|
||||
# UV
|
||||
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
|
||||
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||
# commonly ignored for libraries.
|
||||
#uv.lock
|
||||
|
||||
# poetry
|
||||
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
||||
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||
# commonly ignored for libraries.
|
||||
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
||||
#poetry.lock
|
||||
|
||||
# pdm
|
||||
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
||||
#pdm.lock
|
||||
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
||||
# in version control.
|
||||
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
|
||||
.pdm.toml
|
||||
.pdm-python
|
||||
.pdm-build/
|
||||
|
||||
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
||||
__pypackages__/
|
||||
|
||||
# Celery stuff
|
||||
celerybeat-schedule
|
||||
celerybeat.pid
|
||||
|
||||
# SageMath parsed files
|
||||
*.sage.py
|
||||
|
||||
# Environments
|
||||
.env
|
||||
.venv
|
||||
env/
|
||||
venv/
|
||||
ENV/
|
||||
env.bak/
|
||||
venv.bak/
|
||||
|
||||
# Spyder project settings
|
||||
.spyderproject
|
||||
.spyproject
|
||||
|
||||
# Rope project settings
|
||||
.ropeproject
|
||||
|
||||
# mkdocs documentation
|
||||
/site
|
||||
|
||||
# mypy
|
||||
.mypy_cache/
|
||||
.dmypy.json
|
||||
dmypy.json
|
||||
|
||||
# Pyre type checker
|
||||
.pyre/
|
||||
|
||||
# pytype static type analyzer
|
||||
.pytype/
|
||||
|
||||
# Cython debug symbols
|
||||
cython_debug/
|
||||
|
||||
# PyCharm
|
||||
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
||||
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
||||
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
||||
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
||||
#.idea/
|
||||
|
||||
# PyPI configuration file
|
||||
.pypirc
|
||||
@@ -1,201 +1,201 @@
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
@@ -1,229 +1,167 @@
|
||||
# (Out Dated) ComfyUI-Geeky-LatentSyncWrapper 1.5 (Mediapipe isn't compatible with some changes to ComfyUI, will need to downgrade python version of portable comfyUI and remove xformers).
|
||||
|
||||
Unofficial **optimized and enhanced** fork of [LatentSync 1.5](https://github.com/bytedance/LatentSync) implementation for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) on Windows and WSL 2.0.
|
||||
|
||||
Works with ComfyUI Portable version 3.49 https://github.com/comfyanonymous/ComfyUI/releases/tag/v0.3.49
|
||||
|
||||
This node provides advanced lip-sync capabilities in ComfyUI using ByteDance's LatentSync 1.5 model with **significantly improved performance, memory efficiency, and stability**. This fork focuses on speed, reliability, and conflict-free coexistence with other LatentSync implementations.
|
||||
|
||||
<img width="1893" height="1357" alt="workflow (4)" src="https://github.com/user-attachments/assets/51d3b98a-e549-4c71-a2f2-2af5cc52f291" />
|
||||
|
||||
## Why This Fork? Performance & Stability
|
||||
|
||||
**🚀 Much Faster Performance**: This implementation is significantly faster than other versions and eliminates OOM (Out of Memory) errors that plague other implementations.
|
||||
|
||||
**🧠 Better Memory Management**: Intelligent VRAM usage with user-selectable settings (high/medium/low) and automatic cleanup prevents memory issues.
|
||||
|
||||
**🔒 Conflict-Free**: Can be installed alongside other LatentSync implementations without interference - uses isolated paths and unique node names.
|
||||
|
||||
**⚡ LatentSync 1.5 vs 1.6**: We use LatentSync 1.5 instead of 1.6 because:
|
||||
- **More Stable**: 1.5 has proven stability and reliability in production use
|
||||
- **Better Performance**: 1.5 runs faster and uses less VRAM than 1.6
|
||||
- **No Manual Downloads**: 1.5 models download automatically, unlike 1.6's private repository requirements
|
||||
- **Fewer Dependencies**: Simpler, more reliable dependency chain
|
||||
|
||||
## What's New in This Optimized Fork?
|
||||
|
||||
### Performance Enhancements
|
||||
1. **Advanced Memory Management**: Intelligent VRAM allocation with user-selectable modes
|
||||
2. **Faster Processing**: Optimized batch processing and GPU utilization
|
||||
3. **No OOM Errors**: Comprehensive memory cleanup and management
|
||||
4. **Mixed Precision Support**: Automatic FP16 optimization when beneficial
|
||||
|
||||
### User Experience Improvements
|
||||
5. **Single Image Support**: Process individual images with LatentSync
|
||||
6. **Batch Image Processing**: Process multiple images efficiently
|
||||
7. **Smart Temp Management**: Isolated temporary directories prevent conflicts
|
||||
8. **Better Error Handling**: Robust error recovery and informative messages
|
||||
|
||||
### Compatibility Features
|
||||
9. **Conflict-Free Installation**: Can coexist with ShmuelRonen's implementation
|
||||
10. **Unique Node Names**: "Geeky" prefixed nodes prevent naming conflicts
|
||||
11. **Isolated Model Storage**: Uses `geeky_checkpoints/` directory
|
||||
12. **Automatic Path Management**: Handles compatibility transparently
|
||||
|
||||
## Original LatentSync 1.5 Features
|
||||
|
||||
1. **Temporal Layer Improvements**: Corrected implementation provides significantly improved temporal consistency compared to version 1.0
|
||||
2. **Better Chinese Language Support**: Performance on Chinese videos is substantially improved through additional training data
|
||||
3. **Reduced VRAM Requirements**: Optimized to run on 20GB VRAM (RTX 3090 compatible) through various optimizations:
|
||||
- Gradient checkpointing in U-Net, VAE, SyncNet and VideoMAE
|
||||
- Native PyTorch FlashAttention-2 implementation (no xFormers dependency)
|
||||
- More efficient CUDA cache management
|
||||
- Focused training of temporal and audio cross-attention layers only
|
||||
4. **Code Optimizations**:
|
||||
- Removed dependencies on xFormers and Triton
|
||||
- Upgraded to diffusers 0.32.2
|
||||
|
||||
## Compatibility with Other LatentSync Nodes
|
||||
|
||||
This repository can be installed alongside ShmuelRonen's ComfyUI-LatentSyncWrapper **without conflicts**:
|
||||
|
||||
- ✅ **Different node names**: Geeky nodes use "Geeky" prefix ("Geeky LatentSync 1.5 (Optimized)")
|
||||
- ✅ **Separate checkpoints**: Uses `geeky_checkpoints/` directory
|
||||
- ✅ **Independent models**: Downloads to isolated paths
|
||||
- ✅ **No shared resources**: Completely separate from other LatentSync implementations
|
||||
- ✅ **Isolated temp directories**: Prevents interference with other nodes
|
||||
|
||||
Both repositories can coexist and users can choose which nodes to use based on their performance needs.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Before installing this node, you must install the following in order:
|
||||
|
||||
1. [ComfyUI](https://github.com/comfyanonymous/ComfyUI) installed and working
|
||||
|
||||
2. FFmpeg installed on your system:
|
||||
- Windows: Download from [here](https://github.com/BtbN/FFmpeg-Builds/releases) and add to system PATH
|
||||
|
||||
## Installation
|
||||
|
||||
Only proceed with installation after confirming all prerequisites are installed and working.
|
||||
|
||||
1. Clone this repository into your ComfyUI custom_nodes directory:
|
||||
```bash
|
||||
cd ComfyUI/custom_nodes
|
||||
git clone https://github.com/GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper.git
|
||||
cd ComfyUI-Geeky-LatentSyncWrapper
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
## Required Dependencies
|
||||
```
|
||||
diffusers>=0.32.2
|
||||
transformers
|
||||
huggingface-hub
|
||||
omegaconf
|
||||
einops
|
||||
opencv-python
|
||||
mediapipe
|
||||
face-alignment
|
||||
decord
|
||||
ffmpeg-python
|
||||
safetensors
|
||||
soundfile
|
||||
```
|
||||
|
||||
## Note on Model Downloads
|
||||
|
||||
On first use, the node will **automatically download** required model files from HuggingFace:
|
||||
- LatentSync 1.5 UNet model (~5GB)
|
||||
- Whisper model for audio processing (~1.6GB)
|
||||
- All models download automatically - no manual intervention required
|
||||
- Models are stored in isolated `geeky_checkpoints/` directory
|
||||
|
||||
### Checkpoint Directory Structure
|
||||
|
||||
After successful installation and model download, your checkpoint directory structure will look like this:
|
||||
|
||||
```
|
||||
./geeky_checkpoints/
|
||||
|-- .cache/
|
||||
|-- auxiliary/
|
||||
|-- whisper/
|
||||
| `-- tiny.pt
|
||||
|-- config.json
|
||||
|-- latentsync_unet.pt (~5GB)
|
||||
|-- stable_syncnet.pt (~1.6GB)
|
||||
```
|
||||
|
||||
Make sure all these files are present for proper functionality. The main model files are:
|
||||
- `latentsync_unet.pt`: The primary LatentSync 1.5 model
|
||||
- `stable_syncnet.pt`: The SyncNet model for lip-sync supervision
|
||||
- `whisper/tiny.pt`: The Whisper model for audio processing
|
||||
|
||||
## Usage
|
||||
|
||||
### For Videos:
|
||||
1. Select an input video file with a video loader
|
||||
2. Load an audio file using ComfyUI audio loader
|
||||
3. (Optional) Set a seed value for reproducible results
|
||||
4. (Optional) Adjust the lips_expression parameter to control lip movement intensity
|
||||
5. (Optional) Modify the inference_steps parameter to balance quality and speed
|
||||
6. (Optional) Choose VRAM usage setting based on your GPU
|
||||
7. Connect to the **Geeky LatentSync 1.5 (Optimized)** node
|
||||
8. Run the workflow
|
||||
|
||||
### For Single Images:
|
||||
1. Load a single image using ComfyUI's image loader
|
||||
2. Load an audio file using ComfyUI audio loader
|
||||
3. Connect to the **Geeky LatentSync 1.5 (Optimized)** node
|
||||
4. Adjust parameters as needed
|
||||
5. Run the workflow
|
||||
|
||||
### For Batch Images:
|
||||
1. Load multiple images using ComfyUI's batch image loader or image list to batch node
|
||||
2. Load an audio file using ComfyUI audio loader
|
||||
3. Connect to the **Geeky LatentSync 1.5 (Optimized)** node
|
||||
4. Adjust parameters as needed
|
||||
5. Run the workflow
|
||||
|
||||
The processed video or images will be saved in ComfyUI's output directory.
|
||||
|
||||
### Node Parameters:
|
||||
- `images`: Input image(s) - supports single images, video frames, or batch processing
|
||||
- `audio`: Audio input from ComfyUI audio loader
|
||||
- `seed`: Random seed for reproducible results (default: 1247)
|
||||
- `lips_expression`: Controls the expressiveness of lip movements (default: 1.5)
|
||||
- Higher values (2.0-3.0): More pronounced lip movements, better for expressive speech
|
||||
- Lower values (1.0-1.5): Subtler lip movements, better for calm speech
|
||||
- This parameter affects the model's guidance scale, balancing between natural movement and lip sync accuracy
|
||||
- `inference_steps`: Number of denoising steps during inference (default: 20)
|
||||
- Higher values (30-50): Better quality results but slower processing
|
||||
- Lower values (10-15): Faster processing but potentially lower quality
|
||||
- The default of 20 usually provides a good balance between quality and speed
|
||||
- `vram_usage`: **NEW** - Choose memory usage profile (default: medium)
|
||||
- **High**: Maximum performance, uses 95% VRAM, enables all optimizations
|
||||
- **Medium**: Balanced performance, uses 85% VRAM, good for most users
|
||||
- **Low**: Conservative usage, uses 75% VRAM, for systems with limited memory
|
||||
|
||||
### Available Nodes:
|
||||
- **Geeky LatentSync 1.5 (Optimized)**: Main lip-sync processing node
|
||||
- **Geeky Video Length Adjuster (Fast)**: Utility node for video/audio length matching
|
||||
|
||||
### Tips for Better Results:
|
||||
- **Performance**: Start with "medium" VRAM usage and increase to "high" if you have sufficient GPU memory
|
||||
- **Quality**: For speeches or presentations, try increasing lips_expression to 2.0-2.5
|
||||
- **Efficiency**: For quick previews, use "low" VRAM setting with 10-15 inference steps
|
||||
- **Stability**: This implementation handles single images automatically by duplicating frames to match audio length
|
||||
- **Memory**: The optimized memory management prevents OOM errors even with long audio clips
|
||||
|
||||
## Performance Comparison
|
||||
|
||||
| Feature | This Fork (Geeky) | Original Implementation |
|
||||
|---------|-------------------|------------------------|
|
||||
| OOM Errors | ❌ None | ✅ Frequent |
|
||||
| Processing Speed | 🚀 Much Faster | 🐌 Slower |
|
||||
| Memory Usage | 🧠 Optimized | 💾 High |
|
||||
| VRAM Settings | ✅ 3 Modes | ❌ Fixed |
|
||||
| Conflict-Free | ✅ Yes | ❌ No |
|
||||
| Auto Downloads | ✅ Yes | ⚠️ Manual (1.6) |
|
||||
|
||||
## Known Limitations
|
||||
|
||||
- Works best with clear, frontal face images/videos
|
||||
- Currently does not support anime/cartoon faces
|
||||
- Video should be at 25 FPS (will be automatically converted)
|
||||
- Face should be visible throughout the image/video
|
||||
- Single images are automatically extended to match audio duration
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Common Issues:
|
||||
1. **"Geeky model checkpoints already exist"**: This is normal - models are cached for faster startup
|
||||
2. **Memory errors**: Try lowering VRAM usage setting from high → medium → low
|
||||
3. **Slow performance**: Ensure you're using a CUDA-compatible GPU and try "high" VRAM setting
|
||||
4. **Node not appearing**: Restart ComfyUI after installation and refresh your browser
|
||||
|
||||
## Credits
|
||||
|
||||
This optimized fork is based on:
|
||||
- [LatentSync 1.5](https://github.com/bytedance/LatentSync) by ByteDance Research
|
||||
- [ComfyUI-LatentSyncWrapper](https://github.com/ShmuelRonen/ComfyUI-LatentSyncWrapper) by ShmuelRonen
|
||||
- [ComfyUI](https://github.com/comfyanonymous/ComfyUI)
|
||||
|
||||
Special thanks to the original developers for their groundbreaking work. This fork focuses on performance optimization, memory efficiency, and user experience improvements.
|
||||
|
||||
## License
|
||||
|
||||
This project is licensed under the Apache License 2.0 - see the LICENSE file for details.
|
||||
# ComfyUI-Geeky-LatentSyncWrapper 1.5
|
||||
|
||||
Unofficial enhanced fork of [LatentSync 1.5](https://github.com/bytedance/LatentSync) implementation for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) on Windows and WSL 2.0.
|
||||
|
||||
This node provides advanced lip-sync capabilities in ComfyUI using ByteDance's LatentSync 1.5 model. It allows you to synchronize video lips with audio input with improved temporal consistency and better performance on a wider range of languages. This fork adds support for both single images and batch image processing.
|
||||
|
||||
<img width="883" alt="Screenshot 2025-03-27 082328" src="https://github.com/user-attachments/assets/9cb30dd5-3507-4565-a917-ae0ede1a2e89" />
|
||||
|
||||
|
||||
<img width="748" alt="Screenshot 2025-03-27 082535" src="https://github.com/user-attachments/assets/3fb7a39e-da9e-444c-a43a-15de48fa57a9" />
|
||||
|
||||
|
||||
## What's new in this fork?
|
||||
|
||||
1. **Single Image Support**: Process individual images with LatentSync
|
||||
2. **Batch Image Processing**: Process multiple images in a batch for efficient workflows
|
||||
3. **All original LatentSync 1.5 features**: Enhanced temporal consistency, better language support, and reduced VRAM requirements
|
||||
|
||||
## Original LatentSync 1.5 Features
|
||||
|
||||
1. **Temporal Layer Improvements**: Corrected implementation now provides significantly improved temporal consistency compared to version 1.0
|
||||
2. **Better Chinese Language Support**: Performance on Chinese videos is now substantially improved through additional training data
|
||||
3. **Reduced VRAM Requirements**: Now only requires 20GB VRAM (can run on RTX 3090) through various optimizations:
|
||||
- Gradient checkpointing in U-Net, VAE, SyncNet and VideoMAE
|
||||
- Native PyTorch FlashAttention-2 implementation (no xFormers dependency)
|
||||
- More efficient CUDA cache management
|
||||
- Focused training of temporal and audio cross-attention layers only
|
||||
4. **Code Optimizations**:
|
||||
- Removed dependencies on xFormers and Triton
|
||||
- Upgraded to diffusers 0.32.2
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Before installing this node, you must install the following in order:
|
||||
|
||||
1. [ComfyUI](https://github.com/comfyanonymous/ComfyUI) installed and working
|
||||
|
||||
2. FFmpeg installed on your system:
|
||||
- Windows: Download from [here](https://github.com/BtbN/FFmpeg-Builds/releases) and add to system PATH
|
||||
|
||||
## Installation
|
||||
|
||||
Only proceed with installation after confirming all prerequisites are installed and working.
|
||||
|
||||
1. Clone this repository into your ComfyUI custom_nodes directory:
|
||||
```bash
|
||||
cd ComfyUI/custom_nodes
|
||||
git clone https://github.com/GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper.git
|
||||
cd ComfyUI-Geeky-LatentSyncWrapper
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
## Required Dependencies
|
||||
```
|
||||
diffusers>=0.32.2
|
||||
transformers
|
||||
huggingface-hub
|
||||
omegaconf
|
||||
einops
|
||||
opencv-python
|
||||
mediapipe
|
||||
face-alignment
|
||||
decord
|
||||
ffmpeg-python
|
||||
safetensors
|
||||
soundfile
|
||||
```
|
||||
|
||||
## Note on Model Downloads
|
||||
|
||||
On first use, the node will automatically download required model files from HuggingFace:
|
||||
- LatentSync 1.5 UNet model
|
||||
- Whisper model for audio processing
|
||||
- You can also manually download the models from HuggingFace repo: https://huggingface.co/ByteDance/LatentSync-1.5
|
||||
|
||||
### Checkpoint Directory Structure
|
||||
|
||||
After successful installation and model download, your checkpoint directory structure should look like this:
|
||||
|
||||
```
|
||||
./checkpoints/
|
||||
|-- .cache/
|
||||
|-- auxiliary/
|
||||
|-- whisper/
|
||||
| `-- tiny.pt
|
||||
|-- config.json
|
||||
|-- latentsync_unet.pt (~5GB)
|
||||
|-- stable_syncnet.pt (~1.6GB)
|
||||
```
|
||||
|
||||
Make sure all these files are present for proper functionality. The main model files are:
|
||||
- `latentsync_unet.pt`: The primary LatentSync 1.5 model
|
||||
- `stable_syncnet.pt`: The SyncNet model for lip-sync supervision
|
||||
- `whisper/tiny.pt`: The Whisper model for audio processing
|
||||
|
||||
## Usage
|
||||
|
||||
### For Videos:
|
||||
1. Select an input video file with AceNodes video loader
|
||||
2. Load an audio file using ComfyUI audio loader
|
||||
3. (Optional) Set a seed value for reproducible results
|
||||
4. (Optional) Adjust the lips_expression parameter to control lip movement intensity
|
||||
5. (Optional) Modify the inference_steps parameter to balance quality and speed
|
||||
6. Connect to the LatentSync1.5 node
|
||||
7. Run the workflow
|
||||
|
||||
### For Single Images:
|
||||
1. Load a single image using ComfyUI's image loader
|
||||
2. Load an audio file using ComfyUI audio loader
|
||||
3. Connect to the LatentSync1.5 node with the single image mode enabled
|
||||
4. Adjust parameters as needed
|
||||
5. Run the workflow
|
||||
|
||||
### For Batch Images:
|
||||
1. Load multiple images using ComfyUI's batch image loader or image list to batch node
|
||||
2. Load an audio file using ComfyUI audio loader
|
||||
3. Connect to the LatentSync1.5 node with the batch processing mode enabled
|
||||
4. Adjust parameters as needed
|
||||
5. Run the workflow
|
||||
|
||||
The processed video or images will be saved in ComfyUI's output directory.
|
||||
|
||||
### Node Parameters:
|
||||
- `input_type`: Select between video, single image, or batch images
|
||||
- `video_path`: Path to input video file (for video mode)
|
||||
- `image`: Input single image (for single image mode)
|
||||
- `image_batch`: Input batch of images (for batch image mode)
|
||||
- `audio`: Audio input from ComfyUI audio loader
|
||||
- `seed`: Random seed for reproducible results (default: 1247)
|
||||
- `lips_expression`: Controls the expressiveness of lip movements (default: 1.5)
|
||||
- Higher values (2.0-3.0): More pronounced lip movements, better for expressive speech
|
||||
- Lower values (1.0-1.5): Subtler lip movements, better for calm speech
|
||||
- This parameter affects the model's guidance scale, balancing between natural movement and lip sync accuracy
|
||||
- `inference_steps`: Number of denoising steps during inference (default: 20)
|
||||
- Higher values (30-50): Better quality results but slower processing
|
||||
- Lower values (10-15): Faster processing but potentially lower quality
|
||||
- The default of 20 usually provides a good balance between quality and speed
|
||||
- `batch_size`: Number of images to process at once in batch mode (default: 4)
|
||||
- Higher values may require more VRAM
|
||||
|
||||
### Tips for Better Results:
|
||||
- For speeches or presentations where clear lip movements are important, try increasing the lips_expression value to 2.0-2.5
|
||||
- For casual conversations, the default value of 1.5 usually works well
|
||||
- If lip movements appear unnatural or exaggerated, try lowering the lips_expression value
|
||||
- Different values may work better for different languages and speech patterns
|
||||
- If you need higher quality results and have time to wait, increase inference_steps to 30-50
|
||||
- For quicker previews or less critical applications, reduce inference_steps to 10-15
|
||||
- When processing batch images, adjust batch_size based on your available VRAM
|
||||
|
||||
## Known Limitations
|
||||
|
||||
- Works best with clear, frontal face images/videos
|
||||
- Currently does not support anime/cartoon faces
|
||||
- Video should be at 25 FPS (will be automatically converted)
|
||||
- Face should be visible throughout the image/video
|
||||
- Batch processing may require significant VRAM depending on batch size
|
||||
|
||||
## Credits
|
||||
|
||||
This fork is based on:
|
||||
- [LatentSync 1.5](https://github.com/bytedance/LatentSync) by ByteDance Research
|
||||
- [ComfyUI-LatentSyncWrapper](https://github.com/ShmuelRonen/ComfyUI-LatentSyncWrapper) by ShmuelRonen
|
||||
- [ComfyUI](https://github.com/comfyanonymous/ComfyUI)
|
||||
|
||||
## License
|
||||
|
||||
This project is licensed under the Apache License 2.0 - see the LICENSE file for details.
|
||||
|
||||
+2
-2
@@ -1,3 +1,3 @@
|
||||
from .nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
|
||||
from .nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
|
||||
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS']
|
||||
@@ -3,25 +3,6 @@ import tempfile
|
||||
import uuid
|
||||
import sys
|
||||
import shutil
|
||||
import time
|
||||
|
||||
# Global model cache to avoid reloading models - Using unique name to avoid conflicts
|
||||
_GEEKY_MODEL_CACHE = {}
|
||||
|
||||
# Function to check for potential conflicts with other LatentSync implementations
|
||||
def check_for_conflicts():
|
||||
"""Check if other LatentSync implementations might conflict"""
|
||||
try:
|
||||
import folder_paths
|
||||
custom_nodes_dir = folder_paths.get_folder_paths("custom_nodes")[0]
|
||||
original_path = os.path.join(custom_nodes_dir, "ComfyUI-LatentSyncWrapper")
|
||||
|
||||
if os.path.exists(original_path):
|
||||
print("[Geeky LatentSync] Detected ComfyUI-LatentSyncWrapper - using isolated paths to avoid conflicts")
|
||||
return True
|
||||
except:
|
||||
pass
|
||||
return False
|
||||
|
||||
# Function to find ComfyUI directories
|
||||
def get_comfyui_temp_dir():
|
||||
@@ -90,10 +71,6 @@ def cleanup_comfyui_temp_directories():
|
||||
except Exception as e:
|
||||
print(f"Error cleaning up temp directories: {str(e)}")
|
||||
|
||||
def get_unique_temp_path(suffix=""):
|
||||
"""Generate unique temp paths for geeky wrapper"""
|
||||
return os.path.join(tempfile.gettempdir(), f"geeky_latentsync_{uuid.uuid4().hex[:8]}{suffix}")
|
||||
|
||||
# Create a module-level function to set up system-wide temp directory
|
||||
def init_temp_directories():
|
||||
"""Initialize global temporary directory settings"""
|
||||
@@ -103,19 +80,9 @@ def init_temp_directories():
|
||||
# Generate a unique base directory for this module
|
||||
system_temp = tempfile.gettempdir()
|
||||
unique_id = str(uuid.uuid4())[:8]
|
||||
temp_base_path = os.path.join(system_temp, f"geeky_latentsync_{unique_id}")
|
||||
temp_base_path = os.path.join(system_temp, f"latentsync_{unique_id}")
|
||||
os.makedirs(temp_base_path, exist_ok=True)
|
||||
|
||||
# Create a persistent model cache directory
|
||||
model_cache_dir = os.path.join(system_temp, "geeky_latentsync_model_cache")
|
||||
os.makedirs(model_cache_dir, exist_ok=True)
|
||||
# Link it into our temp directory for convenience
|
||||
try:
|
||||
os.symlink(model_cache_dir, os.path.join(temp_base_path, "model_cache"), target_is_directory=True)
|
||||
except (OSError, NotImplementedError):
|
||||
# If symlinks aren't supported, just use the cache dir directly
|
||||
shutil.copytree(model_cache_dir, os.path.join(temp_base_path, "model_cache"), dirs_exist_ok=True)
|
||||
|
||||
# Override environment variables that control temp directories
|
||||
os.environ['TMPDIR'] = temp_base_path
|
||||
os.environ['TEMP'] = temp_base_path
|
||||
@@ -151,25 +118,13 @@ def init_temp_directories():
|
||||
# Function to clean up everything when the module exits
|
||||
def module_cleanup():
|
||||
"""Clean up all resources when the module is unloaded"""
|
||||
global MODULE_TEMP_DIR, _GEEKY_MODEL_CACHE
|
||||
global MODULE_TEMP_DIR
|
||||
|
||||
# Clear model cache references to free memory
|
||||
_GEEKY_MODEL_CACHE.clear()
|
||||
|
||||
# Clean up temp directory except model cache
|
||||
# Clean up our module temp directory
|
||||
if MODULE_TEMP_DIR and os.path.exists(MODULE_TEMP_DIR):
|
||||
try:
|
||||
for item in os.listdir(MODULE_TEMP_DIR):
|
||||
if item != "model_cache":
|
||||
path = os.path.join(MODULE_TEMP_DIR, item)
|
||||
if os.path.isdir(path):
|
||||
shutil.rmtree(path, ignore_errors=True)
|
||||
else:
|
||||
try:
|
||||
os.remove(path)
|
||||
except:
|
||||
pass
|
||||
print(f"Cleaned up module temp directory (preserving model cache)")
|
||||
shutil.rmtree(MODULE_TEMP_DIR, ignore_errors=True)
|
||||
print(f"Cleaned up module temp directory: {MODULE_TEMP_DIR}")
|
||||
except:
|
||||
pass
|
||||
|
||||
@@ -200,9 +155,6 @@ from PIL import Image
|
||||
from decimal import Decimal, ROUND_UP
|
||||
import requests
|
||||
|
||||
# Check for conflicts with other implementations
|
||||
conflict_detected = check_for_conflicts()
|
||||
|
||||
# Modify folder_paths module to use our temp directory
|
||||
if hasattr(folder_paths, "get_temp_directory"):
|
||||
original_get_temp = folder_paths.get_temp_directory
|
||||
@@ -211,44 +163,12 @@ else:
|
||||
# Add the function if it doesn't exist
|
||||
setattr(folder_paths, 'get_temp_directory', lambda: MODULE_TEMP_DIR)
|
||||
|
||||
def get_cached_model(model_path, model_type, device):
|
||||
"""Load model from geeky-specific cache or disk and cache it"""
|
||||
global _GEEKY_MODEL_CACHE
|
||||
cache_key = f"geeky_{model_type}_{model_path}"
|
||||
|
||||
if cache_key in _GEEKY_MODEL_CACHE:
|
||||
# Check if the cached model is on the right device
|
||||
cached_model = _GEEKY_MODEL_CACHE[cache_key]
|
||||
model_device = next(cached_model.parameters()).device
|
||||
if str(model_device) == str(device):
|
||||
print(f"Using cached {model_type} model from Geeky cache")
|
||||
return cached_model
|
||||
else:
|
||||
print(f"Moving cached {model_type} model to {device}")
|
||||
cached_model = cached_model.to(device)
|
||||
return cached_model
|
||||
|
||||
print(f"Loading {model_type} model from disk into Geeky cache")
|
||||
# Load the model
|
||||
model = torch.load(model_path, map_location=device)
|
||||
|
||||
# Cache the model
|
||||
_GEEKY_MODEL_CACHE[cache_key] = model
|
||||
return model
|
||||
|
||||
def import_inference_script(script_path):
|
||||
"""Import a Python file as a module using its file path."""
|
||||
if not os.path.exists(script_path):
|
||||
raise ImportError(f"Script not found: {script_path}")
|
||||
|
||||
module_name = "geeky_latentsync_inference" # Unique module name
|
||||
|
||||
# Check if the module is already loaded
|
||||
if module_name in sys.modules:
|
||||
print("Using previously imported Geeky inference module")
|
||||
return sys.modules[module_name]
|
||||
|
||||
print(f"Importing Geeky inference script from {script_path}")
|
||||
module_name = "latentsync_inference"
|
||||
spec = importlib.util.spec_from_file_location(module_name, script_path)
|
||||
if spec is None:
|
||||
raise ImportError(f"Failed to create module spec for {script_path}")
|
||||
@@ -294,6 +214,7 @@ def check_ffmpeg():
|
||||
def check_and_install_dependencies():
|
||||
if not check_ffmpeg():
|
||||
raise RuntimeError("FFmpeg is required but not found")
|
||||
|
||||
required_packages = [
|
||||
'omegaconf',
|
||||
'transformers',
|
||||
@@ -303,20 +224,10 @@ def check_and_install_dependencies():
|
||||
'diffusers',
|
||||
'ffmpeg-python'
|
||||
]
|
||||
|
||||
# Check if we've already run this function successfully
|
||||
cache_dir = os.path.join(MODULE_TEMP_DIR, "model_cache")
|
||||
# Create the cache directory if it doesn't exist
|
||||
os.makedirs(cache_dir, exist_ok=True)
|
||||
|
||||
cache_marker = os.path.join(cache_dir, ".geeky_deps_installed")
|
||||
if os.path.exists(cache_marker):
|
||||
print("Geeky dependencies already verified, skipping check.")
|
||||
return
|
||||
|
||||
|
||||
def is_package_installed(package_name):
|
||||
return importlib.util.find_spec(package_name) is not None
|
||||
|
||||
|
||||
def install_package(package):
|
||||
python_exe = sys.executable
|
||||
try:
|
||||
@@ -327,29 +238,15 @@ def check_and_install_dependencies():
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(f"Error installing {package}: {str(e)}")
|
||||
raise RuntimeError(f"Failed to install required package: {package}")
|
||||
|
||||
missing_packages = []
|
||||
|
||||
for package in required_packages:
|
||||
if not is_package_installed(package):
|
||||
missing_packages.append(package)
|
||||
|
||||
if missing_packages:
|
||||
print(f"Installing missing packages: {', '.join(missing_packages)}")
|
||||
for package in missing_packages:
|
||||
print(f"Installing required package: {package}")
|
||||
try:
|
||||
install_package(package)
|
||||
except Exception as e:
|
||||
print(f"Warning: Failed to install {package}: {str(e)}")
|
||||
raise
|
||||
else:
|
||||
print("All required packages are already installed.")
|
||||
|
||||
# Create marker file
|
||||
try:
|
||||
with open(cache_marker, 'w') as f:
|
||||
f.write(f"Geeky dependencies checked on {time.ctime()}")
|
||||
except Exception as e:
|
||||
print(f"Warning: Could not create cache marker file: {str(e)}")
|
||||
|
||||
def normalize_path(path):
|
||||
"""Normalize path to handle spaces and special characters"""
|
||||
@@ -380,24 +277,10 @@ def get_ext_dir(subpath=None, mkdir=False):
|
||||
def download_model(url, save_path):
|
||||
"""Download a model from a URL and save it to the specified path."""
|
||||
os.makedirs(os.path.dirname(save_path), exist_ok=True)
|
||||
# Check if file already exists
|
||||
if os.path.exists(save_path):
|
||||
print(f"Model file already exists at {save_path}, skipping download.")
|
||||
return
|
||||
|
||||
print(f"Downloading {url} to {save_path}...")
|
||||
response = requests.get(url, stream=True)
|
||||
total_size = int(response.headers.get('content-length', 0))
|
||||
downloaded = 0
|
||||
|
||||
with open(save_path, "wb") as f:
|
||||
for chunk in response.iter_content(chunk_size=8192):
|
||||
f.write(chunk)
|
||||
downloaded += len(chunk)
|
||||
if total_size > 0:
|
||||
percent = (downloaded / total_size) * 100
|
||||
print(f"\rDownload progress: {percent:.1f}%", end="")
|
||||
print("\nDownload complete")
|
||||
|
||||
def pre_download_models():
|
||||
"""Pre-download all required models."""
|
||||
@@ -409,23 +292,13 @@ def pre_download_models():
|
||||
cache_dir = os.path.join(MODULE_TEMP_DIR, "model_cache")
|
||||
os.makedirs(cache_dir, exist_ok=True)
|
||||
|
||||
# Check if we've already run this function successfully by creating a marker file
|
||||
cache_marker = os.path.join(cache_dir, ".geeky_cache_complete")
|
||||
if os.path.exists(cache_marker):
|
||||
print("Pre-downloaded Geeky models already exist, skipping download.")
|
||||
return
|
||||
|
||||
for model_name, url in models.items():
|
||||
save_path = os.path.join(cache_dir, model_name)
|
||||
if not os.path.exists(save_path):
|
||||
print(f"Downloading {model_name}...")
|
||||
download_model(url, save_path)
|
||||
else:
|
||||
print(f"{model_name} already exists in Geeky cache.")
|
||||
|
||||
# Create marker file to indicate successful completion
|
||||
with open(cache_marker, 'w') as f:
|
||||
f.write(f"Geeky cache completed on {time.ctime()}")
|
||||
print(f"{model_name} already exists in cache.")
|
||||
|
||||
def setup_models():
|
||||
"""Setup and pre-download all required models."""
|
||||
@@ -435,9 +308,9 @@ def setup_models():
|
||||
# Pre-download additional models
|
||||
pre_download_models()
|
||||
|
||||
# Existing setup logic for LatentSync models - using unique directory
|
||||
# Existing setup logic for LatentSync models
|
||||
cur_dir = get_ext_dir()
|
||||
ckpt_dir = os.path.join(cur_dir, "geeky_checkpoints") # Changed from "checkpoints" to "geeky_checkpoints"
|
||||
ckpt_dir = os.path.join(cur_dir, "checkpoints")
|
||||
whisper_dir = os.path.join(ckpt_dir, "whisper")
|
||||
os.makedirs(ckpt_dir, exist_ok=True)
|
||||
os.makedirs(whisper_dir, exist_ok=True)
|
||||
@@ -449,28 +322,24 @@ def setup_models():
|
||||
unet_path = os.path.join(ckpt_dir, "latentsync_unet.pt")
|
||||
whisper_path = os.path.join(whisper_dir, "tiny.pt")
|
||||
|
||||
# Only download if the files don't already exist
|
||||
if os.path.exists(unet_path) and os.path.exists(whisper_path):
|
||||
print("Geeky model checkpoints already exist, skipping download.")
|
||||
return
|
||||
|
||||
print("Downloading required Geeky model checkpoints... This may take a while.")
|
||||
try:
|
||||
from huggingface_hub import snapshot_download
|
||||
snapshot_download(repo_id="ByteDance/LatentSync-1.5",
|
||||
allow_patterns=["latentsync_unet.pt", "whisper/tiny.pt"],
|
||||
local_dir=ckpt_dir,
|
||||
local_dir_use_symlinks=False,
|
||||
cache_dir=temp_downloads)
|
||||
print("Geeky model checkpoints downloaded successfully!")
|
||||
except Exception as e:
|
||||
print(f"Error downloading Geeky models: {str(e)}")
|
||||
print("\nPlease download models manually for Geeky LatentSync:")
|
||||
print("1. Visit: https://huggingface.co/chunyu-li/LatentSync")
|
||||
print("2. Download: latentsync_unet.pt and whisper/tiny.pt")
|
||||
print(f"3. Place them in: {ckpt_dir}")
|
||||
print(f" with whisper/tiny.pt in: {whisper_dir}")
|
||||
raise RuntimeError("Geeky model download failed. See instructions above.")
|
||||
if not (os.path.exists(unet_path) and os.path.exists(whisper_path)):
|
||||
print("Downloading required model checkpoints... This may take a while.")
|
||||
try:
|
||||
from huggingface_hub import snapshot_download
|
||||
snapshot_download(repo_id="ByteDance/LatentSync-1.5",
|
||||
allow_patterns=["latentsync_unet.pt", "whisper/tiny.pt"],
|
||||
local_dir=ckpt_dir,
|
||||
local_dir_use_symlinks=False,
|
||||
cache_dir=temp_downloads)
|
||||
print("Model checkpoints downloaded successfully!")
|
||||
except Exception as e:
|
||||
print(f"Error downloading models: {str(e)}")
|
||||
print("\nPlease download models manually:")
|
||||
print("1. Visit: https://huggingface.co/chunyu-li/LatentSync")
|
||||
print("2. Download: latentsync_unet.pt and whisper/tiny.pt")
|
||||
print(f"3. Place them in: {ckpt_dir}")
|
||||
print(f" with whisper/tiny.pt in: {whisper_dir}")
|
||||
raise RuntimeError("Model download failed. See instructions above.")
|
||||
|
||||
class GeekyLatentSyncNode:
|
||||
def __init__(self):
|
||||
@@ -480,8 +349,8 @@ class GeekyLatentSyncNode:
|
||||
os.makedirs(MODULE_TEMP_DIR, exist_ok=True)
|
||||
|
||||
# Ensure ComfyUI temp doesn't exist
|
||||
comfyui_temp = get_comfyui_temp_dir()
|
||||
if comfyui_temp and os.path.exists(comfyui_temp):
|
||||
comfyui_temp = "D:\\ComfyUI_windows\\temp"
|
||||
if os.path.exists(comfyui_temp):
|
||||
backup_name = f"{comfyui_temp}_backup_{uuid.uuid4().hex[:8]}"
|
||||
try:
|
||||
os.rename(comfyui_temp, backup_name)
|
||||
@@ -499,7 +368,6 @@ class GeekyLatentSyncNode:
|
||||
"seed": ("INT", {"default": 1247}),
|
||||
"lips_expression": ("FLOAT", {"default": 1.5, "min": 1.0, "max": 3.0, "step": 0.1}),
|
||||
"inference_steps": ("INT", {"default": 20, "min": 1, "max": 999, "step": 1}),
|
||||
"vram_usage": (["high", "medium", "low"], {"default": "medium"}),
|
||||
},}
|
||||
|
||||
CATEGORY = "GeekyLatentSync"
|
||||
@@ -519,66 +387,51 @@ class GeekyLatentSyncNode:
|
||||
processed_batch = processed_batch[..., :3]
|
||||
return processed_batch
|
||||
|
||||
def inference(self, images, audio, seed, lips_expression=1.5, inference_steps=20, vram_usage="medium"):
|
||||
# Add timing information
|
||||
import time
|
||||
start_time = time.time()
|
||||
|
||||
def inference(self, images, audio, seed, lips_expression=1.5, inference_steps=20):
|
||||
# Use our module temp directory
|
||||
global MODULE_TEMP_DIR
|
||||
|
||||
# Define timing checkpoint function
|
||||
def log_timing(step):
|
||||
elapsed = time.time() - start_time
|
||||
print(f"[Geeky {elapsed:.2f}s] {step}")
|
||||
|
||||
log_timing("Starting Geeky inference")
|
||||
|
||||
# Get GPU capabilities and memory
|
||||
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
|
||||
BATCH_SIZE = 4
|
||||
use_mixed_precision = False
|
||||
|
||||
# Set VRAM usage based on user preference
|
||||
if torch.cuda.is_available():
|
||||
gpu_mem = torch.cuda.get_device_properties(0).total_memory
|
||||
# Convert to GB
|
||||
gpu_mem_gb = gpu_mem / (1024 ** 3)
|
||||
|
||||
# Dynamic batch size and settings based on VRAM usage preference
|
||||
if vram_usage == "high":
|
||||
BATCH_SIZE = min(32, 120 // inference_steps)
|
||||
|
||||
# Dynamically adjust batch size based on GPU memory
|
||||
if gpu_mem_gb > 20: # High-end GPUs
|
||||
BATCH_SIZE = 32
|
||||
enable_tf32 = True
|
||||
use_mixed_precision = True
|
||||
torch.backends.cudnn.benchmark = True
|
||||
torch.backends.cuda.matmul.allow_tf32 = True if hasattr(torch.backends.cuda, "matmul") else False
|
||||
torch.backends.cudnn.allow_tf32 = True
|
||||
torch.cuda.set_per_process_memory_fraction(0.95)
|
||||
print(f"Using Geeky high VRAM settings with {BATCH_SIZE} batch size")
|
||||
elif vram_usage == "medium":
|
||||
BATCH_SIZE = min(16, 80 // inference_steps)
|
||||
elif gpu_mem_gb > 8: # Mid-range GPUs
|
||||
BATCH_SIZE = 16
|
||||
enable_tf32 = False
|
||||
use_mixed_precision = True
|
||||
torch.backends.cudnn.benchmark = True
|
||||
torch.cuda.set_per_process_memory_fraction(0.85)
|
||||
print(f"Using Geeky medium VRAM settings with {BATCH_SIZE} batch size")
|
||||
else: # low
|
||||
BATCH_SIZE = min(8, 40 // inference_steps)
|
||||
else: # Lower-end GPUs
|
||||
BATCH_SIZE = 8
|
||||
enable_tf32 = False
|
||||
use_mixed_precision = False
|
||||
torch.cuda.set_per_process_memory_fraction(0.75)
|
||||
print(f"Using Geeky low VRAM settings with {BATCH_SIZE} batch size")
|
||||
|
||||
|
||||
# Set performance options based on GPU capability
|
||||
torch.backends.cudnn.benchmark = True
|
||||
if enable_tf32:
|
||||
torch.backends.cuda.matmul.allow_tf32 = True
|
||||
torch.backends.cudnn.allow_tf32 = True
|
||||
|
||||
# Clear GPU cache before processing
|
||||
torch.cuda.empty_cache()
|
||||
else:
|
||||
# CPU fallback settings
|
||||
BATCH_SIZE = 4
|
||||
print("No GPU detected, using CPU with minimal batch size")
|
||||
|
||||
torch.cuda.set_per_process_memory_fraction(0.8)
|
||||
|
||||
# Create a run-specific subdirectory in our temp directory
|
||||
run_id = ''.join(random.choice("abcdefghijklmnopqrstuvwxyz") for _ in range(5))
|
||||
temp_dir = os.path.join(MODULE_TEMP_DIR, f"geeky_run_{run_id}")
|
||||
temp_dir = os.path.join(MODULE_TEMP_DIR, f"run_{run_id}")
|
||||
os.makedirs(temp_dir, exist_ok=True)
|
||||
|
||||
# Ensure ComfyUI temp doesn't exist again
|
||||
comfyui_temp = get_comfyui_temp_dir()
|
||||
if comfyui_temp and os.path.exists(comfyui_temp):
|
||||
# Ensure ComfyUI temp doesn't exist again (in case something recreated it)
|
||||
comfyui_temp = "D:\\ComfyUI_windows\\temp"
|
||||
if os.path.exists(comfyui_temp):
|
||||
backup_name = f"{comfyui_temp}_backup_{uuid.uuid4().hex[:8]}"
|
||||
try:
|
||||
os.rename(comfyui_temp, backup_name)
|
||||
@@ -591,14 +444,13 @@ class GeekyLatentSyncNode:
|
||||
|
||||
try:
|
||||
# Create temporary file paths in our system temp directory
|
||||
temp_video_path = os.path.join(temp_dir, f"geeky_temp_{run_id}.mp4")
|
||||
output_video_path = os.path.join(temp_dir, f"geeky_latentsync_{run_id}_out.mp4")
|
||||
audio_path = os.path.join(temp_dir, f"geeky_latentsync_{run_id}_audio.wav")
|
||||
temp_video_path = os.path.join(temp_dir, f"temp_{run_id}.mp4")
|
||||
output_video_path = os.path.join(temp_dir, f"latentsync_{run_id}_out.mp4")
|
||||
audio_path = os.path.join(temp_dir, f"latentsync_{run_id}_audio.wav")
|
||||
|
||||
# Get the extension directory
|
||||
cur_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
log_timing("Processing input frames")
|
||||
# Process input frames
|
||||
if isinstance(images, list):
|
||||
frames = torch.stack(images).to(device)
|
||||
@@ -634,9 +486,8 @@ class GeekyLatentSyncNode:
|
||||
single_frame = frames[0]
|
||||
duplicated_frames = single_frame.unsqueeze(0).repeat(required_frames, 1, 1, 1)
|
||||
frames = duplicated_frames
|
||||
print(f"Geeky: Duplicated single image to create {required_frames} frames matching audio duration")
|
||||
print(f"Duplicated single image to create {required_frames} frames matching audio duration")
|
||||
|
||||
log_timing("Processing audio")
|
||||
# Resample audio if needed
|
||||
if sample_rate != 16000:
|
||||
new_sample_rate = 16000
|
||||
@@ -653,7 +504,6 @@ class GeekyLatentSyncNode:
|
||||
"sample_rate": sample_rate
|
||||
}
|
||||
|
||||
log_timing("Saving temporary files")
|
||||
# Move waveform to CPU for saving
|
||||
waveform_cpu = waveform.cpu()
|
||||
torchaudio.save(audio_path, waveform_cpu, sample_rate)
|
||||
@@ -678,19 +528,13 @@ class GeekyLatentSyncNode:
|
||||
packet = stream.encode(None)
|
||||
container.mux(packet)
|
||||
container.close()
|
||||
|
||||
# Free up memory after saving
|
||||
del frames_cpu, waveform_cpu
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
log_timing("Setting up model paths")
|
||||
# Define paths to required files and configs - using geeky_checkpoints
|
||||
# Define paths to required files and configs
|
||||
inference_script_path = os.path.join(cur_dir, "scripts", "inference.py")
|
||||
config_path = os.path.join(cur_dir, "configs", "unet", "stage2.yaml")
|
||||
scheduler_config_path = os.path.join(cur_dir, "configs")
|
||||
ckpt_path = os.path.join(cur_dir, "geeky_checkpoints", "latentsync_unet.pt") # Updated path
|
||||
whisper_ckpt_path = os.path.join(cur_dir, "geeky_checkpoints", "whisper", "tiny.pt") # Updated path
|
||||
ckpt_path = os.path.join(cur_dir, "checkpoints", "latentsync_unet.pt")
|
||||
whisper_ckpt_path = os.path.join(cur_dir, "checkpoints", "whisper", "tiny.pt")
|
||||
|
||||
# Create config and args
|
||||
config = OmegaConf.load(config_path)
|
||||
@@ -728,25 +572,6 @@ class GeekyLatentSyncNode:
|
||||
mask_image_path=mask_image_path
|
||||
)
|
||||
|
||||
# CRITICAL FIX: Create symlink or copy to handle hardcoded paths in inference script
|
||||
old_checkpoints_dir = os.path.join(cur_dir, "checkpoints")
|
||||
if not os.path.exists(old_checkpoints_dir):
|
||||
try:
|
||||
# Try to create a symlink first (faster)
|
||||
os.symlink(os.path.join(cur_dir, "geeky_checkpoints"), old_checkpoints_dir)
|
||||
print(f"Created symlink from {old_checkpoints_dir} to geeky_checkpoints")
|
||||
except (OSError, NotImplementedError):
|
||||
# If symlinks aren't supported, copy the directory
|
||||
shutil.copytree(os.path.join(cur_dir, "geeky_checkpoints"), old_checkpoints_dir, dirs_exist_ok=True)
|
||||
# Create a marker file so we know this is our temporary copy
|
||||
with open(os.path.join(old_checkpoints_dir, ".geeky_temp_copy"), 'w') as f:
|
||||
f.write("Temporary copy created by Geeky LatentSync")
|
||||
print(f"Copied geeky_checkpoints to {old_checkpoints_dir}")
|
||||
elif os.path.isdir(old_checkpoints_dir) and not os.path.islink(old_checkpoints_dir):
|
||||
# If checkpoints directory exists and is not our symlink, warn user
|
||||
print("Warning: Found existing 'checkpoints' directory. This might conflict with other LatentSync implementations.")
|
||||
print("Geeky LatentSync will use the existing directory but recommend using separate installations.")
|
||||
|
||||
# Set PYTHONPATH to include our directories
|
||||
package_root = os.path.dirname(cur_dir)
|
||||
if package_root not in sys.path:
|
||||
@@ -759,14 +584,12 @@ class GeekyLatentSyncNode:
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
# Check and prevent ComfyUI temp creation again
|
||||
comfyui_temp = get_comfyui_temp_dir()
|
||||
if comfyui_temp and os.path.exists(comfyui_temp):
|
||||
if os.path.exists(comfyui_temp):
|
||||
try:
|
||||
os.rename(comfyui_temp, f"{comfyui_temp}_backup_{uuid.uuid4().hex[:8]}")
|
||||
except:
|
||||
pass
|
||||
|
||||
log_timing("Importing inference module")
|
||||
# Import the inference module
|
||||
inference_module = import_inference_script(inference_script_path)
|
||||
|
||||
@@ -778,11 +601,9 @@ class GeekyLatentSyncNode:
|
||||
inference_temp = os.path.join(temp_dir, "temp")
|
||||
os.makedirs(inference_temp, exist_ok=True)
|
||||
|
||||
log_timing("Running Geeky inference")
|
||||
# Run inference
|
||||
inference_module.main(config, args)
|
||||
|
||||
log_timing("Processing output")
|
||||
# Clean GPU cache after inference
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
@@ -792,7 +613,6 @@ class GeekyLatentSyncNode:
|
||||
raise FileNotFoundError(f"Output video not found at: {output_video_path}")
|
||||
|
||||
# Read the processed video - ensure it's loaded as CPU tensor
|
||||
import torchvision.io as io
|
||||
processed_frames = io.read_video(output_video_path, pts_unit='sec')[0]
|
||||
processed_frames = processed_frames.float() / 255.0
|
||||
|
||||
@@ -803,55 +623,39 @@ class GeekyLatentSyncNode:
|
||||
if hasattr(processed_frames, 'device') and processed_frames.device.type == 'cuda':
|
||||
processed_frames = processed_frames.cpu()
|
||||
|
||||
total_time = time.time() - start_time
|
||||
print(f"Geeky total processing time: {total_time:.2f}s")
|
||||
|
||||
return (processed_frames, resampled_audio)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during Geeky inference: {str(e)}")
|
||||
print(f"Error during inference: {str(e)}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
raise
|
||||
|
||||
finally:
|
||||
# Cleanup GPU memory
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
# Clean up the temporary symlink/copy created for inference script compatibility
|
||||
try:
|
||||
old_checkpoints_dir = os.path.join(os.path.dirname(os.path.abspath(__file__)), "checkpoints")
|
||||
if os.path.exists(old_checkpoints_dir):
|
||||
if os.path.islink(old_checkpoints_dir):
|
||||
os.unlink(old_checkpoints_dir)
|
||||
print("Cleaned up temporary symlink")
|
||||
elif os.path.isdir(old_checkpoints_dir):
|
||||
# Only remove if it's our temporary copy (check if it contains geeky files)
|
||||
geeky_marker = os.path.join(old_checkpoints_dir, ".geeky_temp_copy")
|
||||
if os.path.exists(geeky_marker):
|
||||
shutil.rmtree(old_checkpoints_dir)
|
||||
print("Cleaned up temporary copy")
|
||||
except:
|
||||
pass # Ignore cleanup errors
|
||||
|
||||
# Only remove temporary files if successful (keep for debugging if failed)
|
||||
try:
|
||||
# Clean up temporary files individually
|
||||
for path in [temp_video_path, output_video_path, audio_path]:
|
||||
if path and os.path.exists(path):
|
||||
try:
|
||||
os.remove(path)
|
||||
except:
|
||||
pass
|
||||
except:
|
||||
pass # Ignore cleanup errors
|
||||
# Clean up temporary files individually
|
||||
for path in [temp_video_path, output_video_path, audio_path]:
|
||||
if path and os.path.exists(path):
|
||||
try:
|
||||
os.remove(path)
|
||||
print(f"Removed temporary file: {path}")
|
||||
except Exception as e:
|
||||
print(f"Failed to remove {path}: {str(e)}")
|
||||
|
||||
# Remove temporary run directory
|
||||
if temp_dir and os.path.exists(temp_dir):
|
||||
try:
|
||||
shutil.rmtree(temp_dir, ignore_errors=True)
|
||||
print(f"Removed run temporary directory: {temp_dir}")
|
||||
except Exception as e:
|
||||
print(f"Failed to remove temp run directory: {str(e)}")
|
||||
|
||||
# Clean up any ComfyUI temp directories again (in case they were created during execution)
|
||||
cleanup_comfyui_temp_directories()
|
||||
|
||||
# Final GPU cache cleanup
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
|
||||
class GeekyVideoLengthAdjuster:
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
@@ -954,8 +758,8 @@ NODE_CLASS_MAPPINGS = {
|
||||
"GeekyVideoLengthAdjuster": GeekyVideoLengthAdjuster,
|
||||
}
|
||||
|
||||
# Display Names for ComfyUI - Clear distinction from original
|
||||
# Display Names for ComfyUI
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"GeekyLatentSyncNode": "Geeky LatentSync 1.5 (Optimized)",
|
||||
"GeekyVideoLengthAdjuster": "Geeky Video Length Adjuster (Fast)",
|
||||
"GeekyLatentSyncNode": "Geeky LatentSync 1.5",
|
||||
"GeekyVideoLengthAdjuster": "Geeky Video Length Adjuster",
|
||||
}
|
||||
|
||||
@@ -1,85 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import argparse
|
||||
import os
|
||||
from preprocess.affine_transform import affine_transform_multi_gpus
|
||||
from preprocess.remove_broken_videos import remove_broken_videos_multiprocessing
|
||||
from preprocess.detect_shot import detect_shot_multiprocessing
|
||||
from preprocess.filter_high_resolution import filter_high_resolution_multiprocessing
|
||||
from preprocess.resample_fps_hz import resample_fps_hz_multiprocessing
|
||||
from preprocess.segment_videos import segment_videos_multiprocessing
|
||||
from preprocess.sync_av import sync_av_multi_gpus
|
||||
from preprocess.filter_visual_quality import filter_visual_quality_multi_gpus
|
||||
from preprocess.remove_incorrect_affined import remove_incorrect_affined_multiprocessing
|
||||
|
||||
|
||||
def data_processing_pipeline(
|
||||
total_num_workers, per_gpu_num_workers, resolution, sync_conf_threshold, temp_dir, input_dir
|
||||
):
|
||||
print("Removing broken videos...")
|
||||
remove_broken_videos_multiprocessing(input_dir, total_num_workers)
|
||||
|
||||
print("Resampling FPS hz...")
|
||||
resampled_dir = os.path.join(os.path.dirname(input_dir), "resampled")
|
||||
resample_fps_hz_multiprocessing(input_dir, resampled_dir, total_num_workers)
|
||||
|
||||
print("Detecting shot...")
|
||||
shot_dir = os.path.join(os.path.dirname(input_dir), "shot")
|
||||
detect_shot_multiprocessing(resampled_dir, shot_dir, total_num_workers)
|
||||
|
||||
print("Segmenting videos...")
|
||||
segmented_dir = os.path.join(os.path.dirname(input_dir), "segmented")
|
||||
segment_videos_multiprocessing(shot_dir, segmented_dir, total_num_workers)
|
||||
|
||||
# print("Filtering high resolution...")
|
||||
# high_resolution_dir = os.path.join(os.path.dirname(input_dir), "high_resolution")
|
||||
# filter_high_resolution_multiprocessing(segmented_dir, high_resolution_dir, resolution, total_num_workers)
|
||||
|
||||
print("Affine transforming videos...")
|
||||
affine_transformed_dir = os.path.join(os.path.dirname(input_dir), "affine_transformed")
|
||||
affine_transform_multi_gpus(
|
||||
segmented_dir, affine_transformed_dir, temp_dir, resolution, per_gpu_num_workers // 2
|
||||
)
|
||||
|
||||
print("Removing incorrect affined videos...")
|
||||
remove_incorrect_affined_multiprocessing(affine_transformed_dir, total_num_workers)
|
||||
|
||||
print("Syncing audio and video...")
|
||||
av_synced_dir = os.path.join(os.path.dirname(input_dir), f"av_synced_{sync_conf_threshold}")
|
||||
sync_av_multi_gpus(affine_transformed_dir, av_synced_dir, temp_dir, per_gpu_num_workers, sync_conf_threshold)
|
||||
|
||||
print("Filtering visual quality...")
|
||||
high_visual_quality_dir = os.path.join(os.path.dirname(input_dir), "high_visual_quality")
|
||||
filter_visual_quality_multi_gpus(av_synced_dir, high_visual_quality_dir, per_gpu_num_workers)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--total_num_workers", type=int, default=100)
|
||||
parser.add_argument("--per_gpu_num_workers", type=int, default=20)
|
||||
parser.add_argument("--resolution", type=int, default=256)
|
||||
parser.add_argument("--sync_conf_threshold", type=int, default=3)
|
||||
parser.add_argument("--temp_dir", type=str, default="temp")
|
||||
parser.add_argument("--input_dir", type=str, required=True)
|
||||
args = parser.parse_args()
|
||||
|
||||
data_processing_pipeline(
|
||||
args.total_num_workers,
|
||||
args.per_gpu_num_workers,
|
||||
args.resolution,
|
||||
args.sync_conf_threshold,
|
||||
args.temp_dir,
|
||||
args.input_dir,
|
||||
)
|
||||
@@ -1,62 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import tqdm
|
||||
from multiprocessing import Pool
|
||||
|
||||
paths = []
|
||||
|
||||
|
||||
def gather_paths(input_dir, output_dir):
|
||||
for video in sorted(os.listdir(input_dir)):
|
||||
if video.endswith(".mp4"):
|
||||
video_input = os.path.join(input_dir, video)
|
||||
video_output = os.path.join(output_dir, video)
|
||||
if os.path.isfile(video_output):
|
||||
continue
|
||||
paths.append([video_input, output_dir])
|
||||
elif os.path.isdir(os.path.join(input_dir, video)):
|
||||
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
|
||||
|
||||
|
||||
def detect_shot(video_input, output_dir):
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
video = os.path.basename(video_input)[:-4]
|
||||
command = f"scenedetect --quiet -i {video_input} detect-adaptive --threshold 2 split-video --filename '{video}_shot_$SCENE_NUMBER' --output {output_dir}"
|
||||
# command = f"scenedetect --quiet -i {video_input} detect-adaptive --threshold 2 split-video --high-quality --filename '{video}_shot_$SCENE_NUMBER' --output {output_dir}"
|
||||
subprocess.run(command, shell=True)
|
||||
|
||||
|
||||
def multi_run_wrapper(args):
|
||||
return detect_shot(*args)
|
||||
|
||||
|
||||
def detect_shot_multiprocessing(input_dir, output_dir, num_workers):
|
||||
print(f"Recursively gathering video paths of {input_dir} ...")
|
||||
gather_paths(input_dir, output_dir)
|
||||
|
||||
print(f"Detecting shot of {input_dir} ...")
|
||||
with Pool(num_workers) as pool:
|
||||
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
|
||||
pass
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/ads/high-resolution"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/ads/shot"
|
||||
num_workers = 50
|
||||
|
||||
detect_shot_multiprocessing(input_dir, output_dir, num_workers)
|
||||
@@ -1,112 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import mediapipe as mp
|
||||
from latentsync.utils.util import read_video
|
||||
import os
|
||||
import tqdm
|
||||
import shutil
|
||||
from multiprocessing import Pool
|
||||
|
||||
paths = []
|
||||
|
||||
|
||||
def gather_video_paths(input_dir, output_dir, resolution):
|
||||
for video in sorted(os.listdir(input_dir)):
|
||||
if video.endswith(".mp4"):
|
||||
video_input = os.path.join(input_dir, video)
|
||||
video_output = os.path.join(output_dir, video)
|
||||
if os.path.isfile(video_output):
|
||||
continue
|
||||
paths.append([video_input, video_output, resolution])
|
||||
elif os.path.isdir(os.path.join(input_dir, video)):
|
||||
gather_video_paths(os.path.join(input_dir, video), os.path.join(output_dir, video), resolution)
|
||||
|
||||
|
||||
class FaceDetector:
|
||||
def __init__(self, resolution=256):
|
||||
self.face_detection = mp.solutions.face_detection.FaceDetection(
|
||||
model_selection=0, min_detection_confidence=0.5
|
||||
)
|
||||
self.resolution = resolution
|
||||
|
||||
def detect_face(self, image):
|
||||
height, width = image.shape[:2]
|
||||
# Process the image and detect faces.
|
||||
results = self.face_detection.process(image)
|
||||
|
||||
if not results.detections: # Face not detected
|
||||
raise Exception("Face not detected")
|
||||
|
||||
if len(results.detections) != 1:
|
||||
return False
|
||||
detection = results.detections[0] # Only use the first face in the image
|
||||
|
||||
bounding_box = detection.location_data.relative_bounding_box
|
||||
face_width = int(bounding_box.width * width)
|
||||
face_height = int(bounding_box.height * height)
|
||||
if face_width < self.resolution or face_height < self.resolution:
|
||||
return False
|
||||
return True
|
||||
|
||||
def detect_video(self, video_path):
|
||||
video_frames = read_video(video_path, change_fps=False)
|
||||
if len(video_frames) == 0:
|
||||
return False
|
||||
for frame in video_frames:
|
||||
if not self.detect_face(frame):
|
||||
return False
|
||||
return True
|
||||
|
||||
def close(self):
|
||||
self.face_detection.close()
|
||||
|
||||
|
||||
def filter_video(video_input, video_out, resolution):
|
||||
if os.path.isfile(video_out):
|
||||
return
|
||||
face_detector = FaceDetector(resolution)
|
||||
try:
|
||||
save = face_detector.detect_video(video_input)
|
||||
except Exception as e:
|
||||
# print(f"Exception: {e} Input video: {video_input}")
|
||||
face_detector.close()
|
||||
return
|
||||
if save:
|
||||
os.makedirs(os.path.dirname(video_out), exist_ok=True)
|
||||
shutil.copy(video_input, video_out)
|
||||
face_detector.close()
|
||||
|
||||
|
||||
def multi_run_wrapper(args):
|
||||
return filter_video(*args)
|
||||
|
||||
|
||||
def filter_high_resolution_multiprocessing(input_dir, output_dir, resolution, num_workers):
|
||||
print(f"Recursively gathering video paths of {input_dir} ...")
|
||||
gather_video_paths(input_dir, output_dir, resolution)
|
||||
|
||||
print(f"Filtering high resolution videos in {input_dir} ...")
|
||||
with Pool(num_workers) as pool:
|
||||
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
|
||||
pass
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai/lichunyu/HDTF/original/train"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai/lichunyu/HDTF/detected/train"
|
||||
resolution = 256
|
||||
num_workers = 50
|
||||
|
||||
filter_high_resolution_multiprocessing(input_dir, output_dir, resolution, num_workers)
|
||||
@@ -1,129 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import tqdm
|
||||
import torch
|
||||
import torchvision
|
||||
import shutil
|
||||
from multiprocessing import Process
|
||||
import numpy as np
|
||||
from decord import VideoReader
|
||||
from einops import rearrange
|
||||
from eval.hyper_iqa import HyperNet, TargetNet
|
||||
|
||||
|
||||
paths = []
|
||||
|
||||
|
||||
def gather_paths(input_dir, output_dir):
|
||||
# os.makedirs(output_dir, exist_ok=True)
|
||||
|
||||
for video in tqdm.tqdm(sorted(os.listdir(input_dir))):
|
||||
if video.endswith(".mp4"):
|
||||
video_input = os.path.join(input_dir, video)
|
||||
video_output = os.path.join(output_dir, video)
|
||||
if os.path.isfile(video_output):
|
||||
continue
|
||||
paths.append((video_input, video_output))
|
||||
elif os.path.isdir(os.path.join(input_dir, video)):
|
||||
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
|
||||
|
||||
|
||||
def read_video(video_path: str):
|
||||
vr = VideoReader(video_path)
|
||||
first_frame = vr[0].asnumpy()
|
||||
middle_frame = vr[len(vr) // 2].asnumpy()
|
||||
last_frame = vr[-1].asnumpy()
|
||||
vr.seek(0)
|
||||
video_frames = np.stack([first_frame, middle_frame, last_frame], axis=0)
|
||||
video_frames = torch.from_numpy(rearrange(video_frames, "b h w c -> b c h w"))
|
||||
video_frames = video_frames / 255.0
|
||||
return video_frames
|
||||
|
||||
|
||||
def func(paths, device_id):
|
||||
device = f"cuda:{device_id}"
|
||||
|
||||
model_hyper = HyperNet(16, 112, 224, 112, 56, 28, 14, 7).to(device)
|
||||
model_hyper.train(False)
|
||||
|
||||
# load the pre-trained model on the koniq-10k dataset
|
||||
model_hyper.load_state_dict(
|
||||
(torch.load("checkpoints/auxiliary/koniq_pretrained.pkl", map_location=device, weights_only=True))
|
||||
)
|
||||
|
||||
transforms = torchvision.transforms.Compose(
|
||||
[
|
||||
torchvision.transforms.CenterCrop(size=224),
|
||||
torchvision.transforms.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),
|
||||
]
|
||||
)
|
||||
|
||||
for video_input, video_output in paths:
|
||||
try:
|
||||
video_frames = read_video(video_input)
|
||||
video_frames = transforms(video_frames)
|
||||
video_frames = video_frames.clone().detach().to(device)
|
||||
paras = model_hyper(video_frames) # 'paras' contains the network weights conveyed to target network
|
||||
|
||||
# Building target network
|
||||
model_target = TargetNet(paras).to(device)
|
||||
for param in model_target.parameters():
|
||||
param.requires_grad = False
|
||||
|
||||
# Quality prediction
|
||||
pred = model_target(paras["target_in_vec"]) # 'paras['target_in_vec']' is the input to target net
|
||||
|
||||
# quality score ranges from 0-100, a higher score indicates a better quality
|
||||
quality_score = pred.mean().item()
|
||||
print(f"Input video: {video_input}\nVisual quality score: {quality_score:.2f}")
|
||||
|
||||
if quality_score >= 40:
|
||||
os.makedirs(os.path.dirname(video_output), exist_ok=True)
|
||||
shutil.copy(video_input, video_output)
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
|
||||
def split(a, n):
|
||||
k, m = divmod(len(a), n)
|
||||
return (a[i * k + min(i, m) : (i + 1) * k + min(i + 1, m)] for i in range(n))
|
||||
|
||||
|
||||
def filter_visual_quality_multi_gpus(input_dir, output_dir, num_workers):
|
||||
gather_paths(input_dir, output_dir)
|
||||
num_devices = torch.cuda.device_count()
|
||||
if num_devices == 0:
|
||||
raise RuntimeError("No GPUs found")
|
||||
split_paths = list(split(paths, num_workers * num_devices))
|
||||
processes = []
|
||||
|
||||
for i in range(num_devices):
|
||||
for j in range(num_workers):
|
||||
process_index = i * num_workers + j
|
||||
process = Process(target=func, args=(split_paths[process_index], i))
|
||||
process.start()
|
||||
processes.append(process)
|
||||
|
||||
for process in processes:
|
||||
process.join()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/VoxCeleb2/av_synced_high"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/VoxCeleb2/high_visual_quality"
|
||||
num_workers = 20 # How many processes per device
|
||||
|
||||
filter_visual_quality_multi_gpus(input_dir, output_dir, num_workers)
|
||||
@@ -1,43 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
from multiprocessing import Pool
|
||||
import tqdm
|
||||
|
||||
from latentsync.utils.av_reader import AVReader
|
||||
from latentsync.utils.util import gather_video_paths_recursively
|
||||
|
||||
|
||||
def remove_broken_video(video_path):
|
||||
try:
|
||||
AVReader(video_path)
|
||||
except Exception:
|
||||
os.remove(video_path)
|
||||
|
||||
|
||||
def remove_broken_videos_multiprocessing(input_dir, num_workers):
|
||||
video_paths = gather_video_paths_recursively(input_dir)
|
||||
|
||||
print("Removing broken videos...")
|
||||
with Pool(num_workers) as pool:
|
||||
for _ in tqdm.tqdm(pool.imap_unordered(remove_broken_video, video_paths), total=len(video_paths)):
|
||||
pass
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual/affine_transformed"
|
||||
num_workers = 50
|
||||
|
||||
remove_broken_videos_multiprocessing(input_dir, num_workers)
|
||||
@@ -1,81 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import mediapipe as mp
|
||||
from latentsync.utils.util import read_video, gather_video_paths_recursively
|
||||
import os
|
||||
import tqdm
|
||||
from multiprocessing import Pool
|
||||
|
||||
|
||||
class FaceDetector:
|
||||
def __init__(self):
|
||||
self.face_detection = mp.solutions.face_detection.FaceDetection(
|
||||
model_selection=0, min_detection_confidence=0.5
|
||||
)
|
||||
|
||||
def detect_face(self, image):
|
||||
# Process the image and detect faces.
|
||||
results = self.face_detection.process(image)
|
||||
|
||||
if not results.detections: # Face not detected
|
||||
return False
|
||||
|
||||
if len(results.detections) != 1:
|
||||
return False
|
||||
return True
|
||||
|
||||
def detect_video(self, video_path):
|
||||
try:
|
||||
video_frames = read_video(video_path, change_fps=False)
|
||||
except Exception as e:
|
||||
print(f"Exception: {e} - {video_path}")
|
||||
return False
|
||||
if len(video_frames) == 0:
|
||||
return False
|
||||
for frame in video_frames:
|
||||
if not self.detect_face(frame):
|
||||
return False
|
||||
return True
|
||||
|
||||
def close(self):
|
||||
self.face_detection.close()
|
||||
|
||||
|
||||
def remove_incorrect_affined(video_path):
|
||||
if not os.path.isfile(video_path):
|
||||
return
|
||||
face_detector = FaceDetector()
|
||||
has_face = face_detector.detect_video(video_path)
|
||||
if not has_face:
|
||||
os.remove(video_path)
|
||||
print(f"Removed: {video_path}")
|
||||
face_detector.close()
|
||||
|
||||
|
||||
def remove_incorrect_affined_multiprocessing(input_dir, num_workers):
|
||||
video_paths = gather_video_paths_recursively(input_dir)
|
||||
print(f"Total videos: {len(video_paths)}")
|
||||
|
||||
print(f"Removing incorrect affined videos in {input_dir} ...")
|
||||
with Pool(num_workers) as pool:
|
||||
for _ in tqdm.tqdm(pool.imap_unordered(remove_incorrect_affined, video_paths), total=len(video_paths)):
|
||||
pass
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual_dcc/high_visual_quality"
|
||||
num_workers = 50
|
||||
|
||||
remove_incorrect_affined_multiprocessing(input_dir, num_workers)
|
||||
@@ -1,70 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import tqdm
|
||||
from multiprocessing import Pool
|
||||
import cv2
|
||||
|
||||
paths = []
|
||||
|
||||
|
||||
def gather_paths(input_dir, output_dir):
|
||||
for video in sorted(os.listdir(input_dir)):
|
||||
if video.endswith(".mp4"):
|
||||
video_input = os.path.join(input_dir, video)
|
||||
video_output = os.path.join(output_dir, video)
|
||||
if os.path.isfile(video_output):
|
||||
continue
|
||||
paths.append([video_input, video_output])
|
||||
elif os.path.isdir(os.path.join(input_dir, video)):
|
||||
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
|
||||
|
||||
|
||||
def get_video_fps(video_path: str):
|
||||
cam = cv2.VideoCapture(video_path)
|
||||
fps = cam.get(cv2.CAP_PROP_FPS)
|
||||
return fps
|
||||
|
||||
|
||||
def resample_fps_hz(video_input, video_output):
|
||||
os.makedirs(os.path.dirname(video_output), exist_ok=True)
|
||||
if get_video_fps(video_input) == 25:
|
||||
command = f"ffmpeg -loglevel error -y -i {video_input} -c:v copy -ar 16000 -q:a 0 {video_output}"
|
||||
else:
|
||||
command = f"ffmpeg -loglevel error -y -i {video_input} -r 25 -ar 16000 -q:a 0 {video_output}"
|
||||
subprocess.run(command, shell=True)
|
||||
|
||||
|
||||
def multi_run_wrapper(args):
|
||||
return resample_fps_hz(*args)
|
||||
|
||||
|
||||
def resample_fps_hz_multiprocessing(input_dir, output_dir, num_workers):
|
||||
print(f"Recursively gathering video paths of {input_dir} ...")
|
||||
gather_paths(input_dir, output_dir)
|
||||
|
||||
print(f"Resampling FPS and Hz of {input_dir} ...")
|
||||
with Pool(num_workers) as pool:
|
||||
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
|
||||
pass
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/HDTF/segmented/train"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/HDTF/resampled_test"
|
||||
num_workers = 20
|
||||
|
||||
resample_fps_hz_multiprocessing(input_dir, output_dir, num_workers)
|
||||
@@ -1,62 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import tqdm
|
||||
from multiprocessing import Pool
|
||||
|
||||
paths = []
|
||||
|
||||
|
||||
def gather_paths(input_dir, output_dir):
|
||||
for video in sorted(os.listdir(input_dir)):
|
||||
if video.endswith(".mp4"):
|
||||
video_basename = video[:-4]
|
||||
video_input = os.path.join(input_dir, video)
|
||||
video_output = os.path.join(output_dir, f"{video_basename}_%03d.mp4")
|
||||
if os.path.isfile(video_output):
|
||||
continue
|
||||
paths.append([video_input, video_output])
|
||||
elif os.path.isdir(os.path.join(input_dir, video)):
|
||||
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
|
||||
|
||||
|
||||
def segment_video(video_input, video_output):
|
||||
os.makedirs(os.path.dirname(video_output), exist_ok=True)
|
||||
command = f"ffmpeg -loglevel error -y -i {video_input} -map 0 -c:v copy -segment_time 5 -f segment -reset_timestamps 1 -q:a 0 {video_output}"
|
||||
# command = f'ffmpeg -loglevel error -y -i {video_input} -map 0 -segment_time 5 -f segment -reset_timestamps 1 -force_key_frames "expr:gte(t,n_forced*5)" -crf 18 -q:a 0 {video_output}'
|
||||
subprocess.run(command, shell=True)
|
||||
|
||||
|
||||
def multi_run_wrapper(args):
|
||||
return segment_video(*args)
|
||||
|
||||
|
||||
def segment_videos_multiprocessing(input_dir, output_dir, num_workers):
|
||||
print(f"Recursively gathering video paths of {input_dir} ...")
|
||||
gather_paths(input_dir, output_dir)
|
||||
|
||||
print(f"Segmenting videos of {input_dir} ...")
|
||||
with Pool(num_workers) as pool:
|
||||
for _ in tqdm.tqdm(pool.imap_unordered(multi_run_wrapper, paths), total=len(paths)):
|
||||
pass
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/avatars_new/cut"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/avatars_new/segmented"
|
||||
num_workers = 50
|
||||
|
||||
segment_videos_multiprocessing(input_dir, output_dir, num_workers)
|
||||
@@ -1,114 +0,0 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import tqdm
|
||||
from eval.syncnet import SyncNetEval
|
||||
from eval.syncnet_detect import SyncNetDetector
|
||||
from eval.eval_sync_conf import syncnet_eval
|
||||
import torch
|
||||
import subprocess
|
||||
import shutil
|
||||
from multiprocessing import Process
|
||||
|
||||
paths = []
|
||||
|
||||
|
||||
def gather_paths(input_dir, output_dir):
|
||||
# os.makedirs(output_dir, exist_ok=True)
|
||||
|
||||
for video in tqdm.tqdm(sorted(os.listdir(input_dir))):
|
||||
if video.endswith(".mp4"):
|
||||
video_input = os.path.join(input_dir, video)
|
||||
video_output = os.path.join(output_dir, video)
|
||||
if os.path.isfile(video_output):
|
||||
continue
|
||||
paths.append((video_input, video_output))
|
||||
elif os.path.isdir(os.path.join(input_dir, video)):
|
||||
gather_paths(os.path.join(input_dir, video), os.path.join(output_dir, video))
|
||||
|
||||
|
||||
def adjust_offset(video_input: str, video_output: str, av_offset: int, fps: int = 25):
|
||||
command = f"ffmpeg -loglevel error -y -i {video_input} -itsoffset {av_offset/fps} -i {video_input} -map 0:v -map 1:a -c copy -q:v 0 -q:a 0 {video_output}"
|
||||
subprocess.run(command, shell=True)
|
||||
|
||||
|
||||
def func(sync_conf_threshold, paths, device_id, process_temp_dir):
|
||||
os.makedirs(process_temp_dir, exist_ok=True)
|
||||
device = f"cuda:{device_id}"
|
||||
|
||||
syncnet = SyncNetEval(device=device)
|
||||
syncnet.loadParameters("checkpoints/auxiliary/syncnet_v2.model")
|
||||
|
||||
detect_results_dir = os.path.join(process_temp_dir, "detect_results")
|
||||
syncnet_eval_results_dir = os.path.join(process_temp_dir, "syncnet_eval_results")
|
||||
|
||||
syncnet_detector = SyncNetDetector(device=device, detect_results_dir=detect_results_dir)
|
||||
|
||||
for video_input, video_output in paths:
|
||||
try:
|
||||
av_offset, conf = syncnet_eval(
|
||||
syncnet, syncnet_detector, video_input, syncnet_eval_results_dir, detect_results_dir
|
||||
)
|
||||
|
||||
if conf >= sync_conf_threshold and abs(av_offset) <= 6:
|
||||
os.makedirs(os.path.dirname(video_output), exist_ok=True)
|
||||
if av_offset == 0:
|
||||
shutil.copy(video_input, video_output)
|
||||
else:
|
||||
adjust_offset(video_input, video_output, av_offset)
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
|
||||
def split(a, n):
|
||||
k, m = divmod(len(a), n)
|
||||
return (a[i * k + min(i, m) : (i + 1) * k + min(i + 1, m)] for i in range(n))
|
||||
|
||||
|
||||
def sync_av_multi_gpus(input_dir, output_dir, temp_dir, num_workers, sync_conf_threshold):
|
||||
gather_paths(input_dir, output_dir)
|
||||
num_devices = torch.cuda.device_count()
|
||||
if num_devices == 0:
|
||||
raise RuntimeError("No GPUs found")
|
||||
split_paths = list(split(paths, num_workers * num_devices))
|
||||
processes = []
|
||||
|
||||
for i in range(num_devices):
|
||||
for j in range(num_workers):
|
||||
process_index = i * num_workers + j
|
||||
process = Process(
|
||||
target=func,
|
||||
args=(
|
||||
sync_conf_threshold,
|
||||
split_paths[process_index],
|
||||
i,
|
||||
os.path.join(temp_dir, f"process_{process_index}"),
|
||||
),
|
||||
)
|
||||
process.start()
|
||||
processes.append(process)
|
||||
|
||||
for process in processes:
|
||||
process.join()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
input_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/ads/affine_transformed"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/VoxCeleb2/temp"
|
||||
temp_dir = "temp"
|
||||
num_workers = 20 # How many processes per device
|
||||
sync_conf_threshold = 3
|
||||
|
||||
sync_av_multi_gpus(input_dir, output_dir, temp_dir, num_workers, sync_conf_threshold)
|
||||
+113
-113
@@ -1,113 +1,113 @@
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
import pandas as pd
|
||||
from tqdm import tqdm
|
||||
|
||||
"""
|
||||
To use this python file, first install yt-dlp by:
|
||||
|
||||
pip install yt-dlp==2024.5.27
|
||||
"""
|
||||
|
||||
|
||||
def download_video(video_url, video_path):
|
||||
get_video_channel_command = f"yt-dlp --print channel {video_url}"
|
||||
result = subprocess.run(get_video_channel_command, shell=True, capture_output=True, text=True)
|
||||
channel = result.stdout.strip()
|
||||
if channel in unwanted_channels:
|
||||
return
|
||||
download_video_command = f"yt-dlp -f bestvideo+bestaudio --skip-unavailable-fragments --merge-output-format mp4 '{video_url}' --output '{video_path}' --external-downloader aria2c --external-downloader-args '-x 16 -k 1M'"
|
||||
try:
|
||||
subprocess.run(download_video_command, shell=True) # ignore_security_alert_wait_for_fix RCE
|
||||
except KeyboardInterrupt:
|
||||
print("Stopped")
|
||||
exit()
|
||||
except:
|
||||
print(f"Error downloading video {video_url}")
|
||||
|
||||
|
||||
def download_videos(num_workers, video_urls, video_paths):
|
||||
with ThreadPoolExecutor(max_workers=num_workers) as executor:
|
||||
executor.map(download_video, video_urls, video_paths)
|
||||
|
||||
|
||||
def read_video_urls(csv_file_path: str, language_column, video_url_column):
|
||||
video_urls = []
|
||||
print("Reading video urls...")
|
||||
df = pd.read_csv(csv_file_path, sep=",")
|
||||
for row in tqdm(df.itertuples(), total=len(df)):
|
||||
language = getattr(row, language_column)
|
||||
video_url = getattr(row, video_url_column)
|
||||
if "clip" in video_url:
|
||||
continue
|
||||
video_urls.append((language, video_url))
|
||||
return video_urls
|
||||
|
||||
|
||||
def extract_vid(video_url):
|
||||
if "watch?v=" in video_url: # ignore_security_alert_wait_for_fix RCE
|
||||
return video_url.split("watch?v=")[1][:11]
|
||||
elif "shorts/" in video_url:
|
||||
return video_url.split("shorts/")[1][:11]
|
||||
elif "youtu.be/" in video_url:
|
||||
return video_url.split("youtu.be/")[1][:11]
|
||||
elif "&v=" in video_url:
|
||||
return video_url.split("&v=")[1][:11]
|
||||
else:
|
||||
print(f"Invalid video url: {video_url}")
|
||||
return None
|
||||
|
||||
|
||||
def main(csv_file_path, language_column, video_url_column, output_dir, num_workers):
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
all_video_urls = read_video_urls(csv_file_path, language_column, video_url_column)
|
||||
|
||||
video_paths = []
|
||||
video_urls = []
|
||||
|
||||
print("Extracting vid...")
|
||||
for language, video_url in tqdm(all_video_urls):
|
||||
vid = extract_vid(video_url)
|
||||
if vid is None:
|
||||
continue
|
||||
video_path = os.path.join(output_dir, language.lower(), f"vid_{vid}.mp4")
|
||||
if os.path.isfile(video_path):
|
||||
continue
|
||||
os.makedirs(os.path.dirname(video_path), exist_ok=True)
|
||||
video_paths.append(video_path)
|
||||
video_urls.append(video_url)
|
||||
|
||||
if len(video_paths) == 0:
|
||||
print("All videos have been downloaded")
|
||||
exit()
|
||||
else:
|
||||
print(f"Downloading {len(video_paths)} videos")
|
||||
|
||||
download_videos(num_workers, video_urls, video_paths)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
csv_file_path = "dcc.csv"
|
||||
language_column = "video_language"
|
||||
video_url_column = "video_link"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual/raw"
|
||||
num_workers = 50
|
||||
|
||||
unwanted_channels = ["TEDx Talks", "DaePyeong Mukbang", "Joeman"]
|
||||
|
||||
main(csv_file_path, language_column, video_url_column, output_dir, num_workers)
|
||||
# Copyright (c) 2024 Bytedance Ltd. and/or its affiliates
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
import pandas as pd
|
||||
from tqdm import tqdm
|
||||
|
||||
"""
|
||||
To use this python file, first install yt-dlp by:
|
||||
|
||||
pip install yt-dlp==2024.5.27
|
||||
"""
|
||||
|
||||
|
||||
def download_video(video_url, video_path):
|
||||
get_video_channel_command = f"yt-dlp --print channel {video_url}"
|
||||
result = subprocess.run(get_video_channel_command, shell=True, capture_output=True, text=True)
|
||||
channel = result.stdout.strip()
|
||||
if channel in unwanted_channels:
|
||||
return
|
||||
download_video_command = f"yt-dlp -f bestvideo+bestaudio --skip-unavailable-fragments --merge-output-format mp4 '{video_url}' --output '{video_path}' --external-downloader aria2c --external-downloader-args '-x 16 -k 1M'"
|
||||
try:
|
||||
subprocess.run(download_video_command, shell=True) # ignore_security_alert_wait_for_fix RCE
|
||||
except KeyboardInterrupt:
|
||||
print("Stopped")
|
||||
exit()
|
||||
except:
|
||||
print(f"Error downloading video {video_url}")
|
||||
|
||||
|
||||
def download_videos(num_workers, video_urls, video_paths):
|
||||
with ThreadPoolExecutor(max_workers=num_workers) as executor:
|
||||
executor.map(download_video, video_urls, video_paths)
|
||||
|
||||
|
||||
def read_video_urls(csv_file_path: str, language_column, video_url_column):
|
||||
video_urls = []
|
||||
print("Reading video urls...")
|
||||
df = pd.read_csv(csv_file_path, sep=",")
|
||||
for row in tqdm(df.itertuples(), total=len(df)):
|
||||
language = getattr(row, language_column)
|
||||
video_url = getattr(row, video_url_column)
|
||||
if "clip" in video_url:
|
||||
continue
|
||||
video_urls.append((language, video_url))
|
||||
return video_urls
|
||||
|
||||
|
||||
def extract_vid(video_url):
|
||||
if "watch?v=" in video_url: # ignore_security_alert_wait_for_fix RCE
|
||||
return video_url.split("watch?v=")[1][:11]
|
||||
elif "shorts/" in video_url:
|
||||
return video_url.split("shorts/")[1][:11]
|
||||
elif "youtu.be/" in video_url:
|
||||
return video_url.split("youtu.be/")[1][:11]
|
||||
elif "&v=" in video_url:
|
||||
return video_url.split("&v=")[1][:11]
|
||||
else:
|
||||
print(f"Invalid video url: {video_url}")
|
||||
return None
|
||||
|
||||
|
||||
def main(csv_file_path, language_column, video_url_column, output_dir, num_workers):
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
all_video_urls = read_video_urls(csv_file_path, language_column, video_url_column)
|
||||
|
||||
video_paths = []
|
||||
video_urls = []
|
||||
|
||||
print("Extracting vid...")
|
||||
for language, video_url in tqdm(all_video_urls):
|
||||
vid = extract_vid(video_url)
|
||||
if vid is None:
|
||||
continue
|
||||
video_path = os.path.join(output_dir, language.lower(), f"vid_{vid}.mp4")
|
||||
if os.path.isfile(video_path):
|
||||
continue
|
||||
os.makedirs(os.path.dirname(video_path), exist_ok=True)
|
||||
video_paths.append(video_path)
|
||||
video_urls.append(video_url)
|
||||
|
||||
if len(video_paths) == 0:
|
||||
print("All videos have been downloaded")
|
||||
exit()
|
||||
else:
|
||||
print(f"Downloading {len(video_paths)} videos")
|
||||
|
||||
download_videos(num_workers, video_urls, video_paths)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
csv_file_path = "dcc.csv"
|
||||
language_column = "video_language"
|
||||
video_url_column = "video_link"
|
||||
output_dir = "/mnt/bn/maliva-gen-ai-v2/chunyu.li/multilingual/raw"
|
||||
num_workers = 50
|
||||
|
||||
unwanted_channels = ["TEDx Talks", "DaePyeong Mukbang", "Joeman"]
|
||||
|
||||
main(csv_file_path, language_column, video_url_column, output_dir, num_workers)
|
||||
|
||||
@@ -0,0 +1,273 @@
|
||||
{
|
||||
"last_node_id": 5,
|
||||
"last_link_id": 5,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 4,
|
||||
"type": "VHS_VideoCombine",
|
||||
"pos": [
|
||||
-877.4164428710938,
|
||||
130.14813232421875
|
||||
],
|
||||
"size": [
|
||||
290.8520202636719,
|
||||
709.1360473632812
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 3
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"shape": 7,
|
||||
"link": 4
|
||||
},
|
||||
{
|
||||
"name": "meta_batch",
|
||||
"type": "VHS_BatchManager",
|
||||
"shape": 7,
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"shape": 7,
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "Filenames",
|
||||
"type": "VHS_FILENAMES",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "comfyui-videohelpersuite",
|
||||
"ver": "1.5.9",
|
||||
"Node name for S&R": "VHS_VideoCombine"
|
||||
},
|
||||
"widgets_values": {
|
||||
"frame_rate": 25,
|
||||
"loop_count": 0,
|
||||
"filename_prefix": "LatentSync",
|
||||
"format": "video/nvenc_h264-mp4",
|
||||
"pix_fmt": "yuv420p",
|
||||
"bitrate": 10,
|
||||
"megabit": true,
|
||||
"save_metadata": true,
|
||||
"pingpong": false,
|
||||
"save_output": true,
|
||||
"videopreview": {
|
||||
"hidden": false,
|
||||
"paused": false,
|
||||
"params": {
|
||||
"filename": "LatentSync_00004-audio.mp4",
|
||||
"subfolder": "",
|
||||
"type": "output",
|
||||
"format": "video/nvenc_h264-mp4",
|
||||
"frame_rate": 25,
|
||||
"workflow": "LatentSync_00004.png",
|
||||
"fullpath": "C:\\Users\\wgray\\Documents\\ComfyUI test branch\\ComfyUI_windows_portable\\ComfyUI\\output\\LatentSync_00004-audio.mp4"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "GeekyKokoroTTS",
|
||||
"pos": [
|
||||
-1664.339599609375,
|
||||
132.1678924560547
|
||||
],
|
||||
"size": [
|
||||
400,
|
||||
252
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
2
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "text_processed",
|
||||
"type": "STRING",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "ComfyUI-Geeky-Kokoro-TTS",
|
||||
"ver": "443f4b0e1c503b47737b18c89f5067b9a6187038",
|
||||
"Node name for S&R": "GeekyKokoroTTS"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Welcome to ComfyUI",
|
||||
"🇺🇸 🚺 Heart ❤️",
|
||||
1,
|
||||
true,
|
||||
false,
|
||||
"🇺🇸 🚺 Sarah",
|
||||
0.5
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "LatentSyncNode",
|
||||
"pos": [
|
||||
-1224.6712646484375,
|
||||
134.58547973632812
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
150
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 5
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"link": 2
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
3
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
4
|
||||
],
|
||||
"slot_index": 1
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"aux_id": "GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper",
|
||||
"ver": "d4c98e2a3718dba91fc87098640db6fea8d86409",
|
||||
"Node name for S&R": "LatentSyncNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
743,
|
||||
"randomize",
|
||||
1.5,
|
||||
20
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-2015.361083984375,
|
||||
147.8274688720703
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
5
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "comfy-core",
|
||||
"ver": "0.3.26",
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"ComfyUI_00611_.png",
|
||||
"image"
|
||||
]
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
2,
|
||||
1,
|
||||
0,
|
||||
3,
|
||||
1,
|
||||
"AUDIO"
|
||||
],
|
||||
[
|
||||
3,
|
||||
3,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
4,
|
||||
3,
|
||||
1,
|
||||
4,
|
||||
1,
|
||||
"AUDIO"
|
||||
],
|
||||
[
|
||||
5,
|
||||
5,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"IMAGE"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.9646149645000188,
|
||||
"offset": [
|
||||
2123.341499155334,
|
||||
21.38434610981646
|
||||
]
|
||||
},
|
||||
"VHS_latentpreview": false,
|
||||
"VHS_latentpreviewrate": 0,
|
||||
"VHS_MetadataImage": true,
|
||||
"VHS_KeepIntermediate": true
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
+311
-349
@@ -1,349 +1,311 @@
|
||||
{
|
||||
"id": "0bed89c3-ac44-45f4-b651-66c7cec8527b",
|
||||
"revision": 0,
|
||||
"last_node_id": 5,
|
||||
"last_link_id": 4,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 1,
|
||||
"type": "GeekyLatentSyncNode",
|
||||
"pos": [
|
||||
4711.150390625,
|
||||
1535.1771240234375
|
||||
],
|
||||
"size": [
|
||||
313.8853454589844,
|
||||
174
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 1
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"link": 2
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
3
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
4
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "ComfyUI-LatentSyncWrapper",
|
||||
"ver": "c467ec665a7bfad3fa3052134fbddb33f8fc58ad",
|
||||
"Node name for S&R": "GeekyLatentSyncNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
312,
|
||||
"randomize",
|
||||
1.8,
|
||||
20,
|
||||
"medium"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "VHS_LoadAudioUpload",
|
||||
"pos": [
|
||||
4716.10595703125,
|
||||
1338.88330078125
|
||||
],
|
||||
"size": [
|
||||
243.818359375,
|
||||
130
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
2
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "comfyui-videohelpersuite",
|
||||
"ver": "1.5.9",
|
||||
"Node name for S&R": "VHS_LoadAudioUpload"
|
||||
},
|
||||
"widgets_values": {
|
||||
"audio": "ComfyUI_temp_efoos_00005_.flac",
|
||||
"start_time": 0,
|
||||
"duration": 0,
|
||||
"choose audio to upload": "image"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "VHS_LoadVideo",
|
||||
"pos": [
|
||||
4438.70849609375,
|
||||
1317.5367431640625
|
||||
],
|
||||
"size": [
|
||||
247.455078125,
|
||||
494.59130859375
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "meta_batch",
|
||||
"shape": 7,
|
||||
"type": "VHS_BatchManager",
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"shape": 7,
|
||||
"type": "VAE",
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "frame_count",
|
||||
"type": "INT",
|
||||
"links": null
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": null
|
||||
},
|
||||
{
|
||||
"name": "video_info",
|
||||
"type": "VHS_VIDEOINFO",
|
||||
"links": []
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "comfyui-videohelpersuite",
|
||||
"ver": "1.5.9",
|
||||
"Node name for S&R": "VHS_LoadVideo"
|
||||
},
|
||||
"widgets_values": {
|
||||
"video": "Upscaled_Wan_00004.mp4",
|
||||
"force_rate": 25,
|
||||
"custom_width": 0,
|
||||
"custom_height": 0,
|
||||
"frame_load_cap": 0,
|
||||
"skip_first_frames": 0,
|
||||
"select_every_nth": 1,
|
||||
"format": "AnimateDiff",
|
||||
"choose video to upload": "image",
|
||||
"videopreview": {
|
||||
"hidden": false,
|
||||
"paused": false,
|
||||
"params": {
|
||||
"filename": "Upscaled_Wan_00004.mp4",
|
||||
"type": "input",
|
||||
"format": "video/mp4",
|
||||
"force_rate": 25,
|
||||
"custom_width": 0,
|
||||
"custom_height": 0,
|
||||
"frame_load_cap": 0,
|
||||
"skip_first_frames": 0,
|
||||
"select_every_nth": 1
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "VHS_VideoCombine",
|
||||
"pos": [
|
||||
5047.68896484375,
|
||||
1320.9598388671875
|
||||
],
|
||||
"size": [
|
||||
450.99688720703125,
|
||||
671.2476806640625
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 3
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"shape": 7,
|
||||
"type": "AUDIO",
|
||||
"link": 4
|
||||
},
|
||||
{
|
||||
"name": "meta_batch",
|
||||
"shape": 7,
|
||||
"type": "VHS_BatchManager",
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"shape": 7,
|
||||
"type": "VAE",
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "Filenames",
|
||||
"type": "VHS_FILENAMES",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "comfyui-videohelpersuite",
|
||||
"ver": "1.5.9",
|
||||
"Node name for S&R": "VHS_VideoCombine"
|
||||
},
|
||||
"widgets_values": {
|
||||
"frame_rate": 25,
|
||||
"loop_count": 0,
|
||||
"filename_prefix": "LatentSync",
|
||||
"format": "video/h264-mp4",
|
||||
"pix_fmt": "yuv420p",
|
||||
"crf": 19,
|
||||
"save_metadata": true,
|
||||
"trim_to_audio": false,
|
||||
"pingpong": false,
|
||||
"save_output": false,
|
||||
"videopreview": {
|
||||
"hidden": false,
|
||||
"paused": false,
|
||||
"params": {
|
||||
"filename": "LatentSync_00003-audio.mp4",
|
||||
"subfolder": "",
|
||||
"type": "temp",
|
||||
"format": "video/h264-mp4",
|
||||
"frame_rate": 25,
|
||||
"workflow": "LatentSync_00003.png",
|
||||
"fullpath": "C:\\Users\\wgray\\AppData\\Local\\Temp\\geeky_latentsync_e108de4d\\latentsync_43dc5e26\\latentsync_76385145\\geeky_latentsync_5bfbc51d\\latentsync_7c3d9782\\latentsync_a2ca1c39\\LatentSync_00003-audio.mp4"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "Note",
|
||||
"pos": [
|
||||
4436.41162109375,
|
||||
1878.3404541015625
|
||||
],
|
||||
"size": [
|
||||
395.8923034667969,
|
||||
100.62284851074219
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"LatentSync can be found with comfyUI manager or here for a faster 1.5 version\n\nhttps://github.com/GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper.git"
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
1,
|
||||
4,
|
||||
0,
|
||||
1,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
2,
|
||||
3,
|
||||
0,
|
||||
1,
|
||||
1,
|
||||
"AUDIO"
|
||||
],
|
||||
[
|
||||
3,
|
||||
1,
|
||||
0,
|
||||
5,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
4,
|
||||
1,
|
||||
1,
|
||||
5,
|
||||
1,
|
||||
"AUDIO"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "Lip Sync Pass",
|
||||
"bounding": [
|
||||
4405.30322265625,
|
||||
1221.0234375,
|
||||
1110.41650390625,
|
||||
783.9459228515625
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ue_links": [],
|
||||
"ds": {
|
||||
"scale": 0.8769226950000164,
|
||||
"offset": [
|
||||
-4066.528439416358,
|
||||
-1092.0980920410036
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.23.4",
|
||||
"VHS_latentpreview": false,
|
||||
"VHS_latentpreviewrate": 0,
|
||||
"VHS_MetadataImage": true,
|
||||
"VHS_KeepIntermediate": true
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
{
|
||||
"last_node_id": 4,
|
||||
"last_link_id": 4,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 3,
|
||||
"type": "LatentSyncNode",
|
||||
"pos": [
|
||||
-1224.6712646484375,
|
||||
134.58547973632812
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
150
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 1
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"link": 2
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
3
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
4
|
||||
],
|
||||
"slot_index": 1
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"aux_id": "GeekyGhost/ComfyUI-Geeky-LatentSyncWrapper",
|
||||
"ver": "d4c98e2a3718dba91fc87098640db6fea8d86409",
|
||||
"Node name for S&R": "LatentSyncNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
991,
|
||||
"randomize",
|
||||
1.5,
|
||||
20
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "VHS_VideoCombine",
|
||||
"pos": [
|
||||
-877.4164428710938,
|
||||
130.14813232421875
|
||||
],
|
||||
"size": [
|
||||
290.8520202636719,
|
||||
334
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 3
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"shape": 7,
|
||||
"link": 4
|
||||
},
|
||||
{
|
||||
"name": "meta_batch",
|
||||
"type": "VHS_BatchManager",
|
||||
"shape": 7,
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"shape": 7,
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "Filenames",
|
||||
"type": "VHS_FILENAMES",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "comfyui-videohelpersuite",
|
||||
"ver": "1.5.9",
|
||||
"Node name for S&R": "VHS_VideoCombine"
|
||||
},
|
||||
"widgets_values": {
|
||||
"frame_rate": 25,
|
||||
"loop_count": 0,
|
||||
"filename_prefix": "LatentSync",
|
||||
"format": "video/nvenc_h264-mp4",
|
||||
"pix_fmt": "yuv420p",
|
||||
"bitrate": 10,
|
||||
"megabit": true,
|
||||
"save_metadata": true,
|
||||
"pingpong": false,
|
||||
"save_output": true,
|
||||
"videopreview": {
|
||||
"hidden": false,
|
||||
"paused": false,
|
||||
"params": {}
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "VHS_LoadVideo",
|
||||
"pos": [
|
||||
-1940.4915771484375,
|
||||
144.70785522460938
|
||||
],
|
||||
"size": [
|
||||
247.455078125,
|
||||
627.8657836914062
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "meta_batch",
|
||||
"type": "VHS_BatchManager",
|
||||
"shape": 7,
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"shape": 7,
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
1
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "frame_count",
|
||||
"type": "INT",
|
||||
"links": null
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": null
|
||||
},
|
||||
{
|
||||
"name": "video_info",
|
||||
"type": "VHS_VIDEOINFO",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "comfyui-videohelpersuite",
|
||||
"ver": "1.5.9",
|
||||
"Node name for S&R": "VHS_LoadVideo"
|
||||
},
|
||||
"widgets_values": {
|
||||
"video": "AnimateDiff_00381.mp4",
|
||||
"force_rate": 0,
|
||||
"custom_width": 0,
|
||||
"custom_height": 0,
|
||||
"frame_load_cap": 0,
|
||||
"skip_first_frames": 0,
|
||||
"select_every_nth": 1,
|
||||
"format": "AnimateDiff",
|
||||
"choose video to upload": "image",
|
||||
"videopreview": {
|
||||
"hidden": false,
|
||||
"paused": false,
|
||||
"params": {
|
||||
"filename": "AnimateDiff_00381.mp4",
|
||||
"type": "input",
|
||||
"format": "video/mp4",
|
||||
"force_rate": 0,
|
||||
"custom_width": 0,
|
||||
"custom_height": 0,
|
||||
"frame_load_cap": 0,
|
||||
"skip_first_frames": 0,
|
||||
"select_every_nth": 1
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "GeekyKokoroTTS",
|
||||
"pos": [
|
||||
-1664.339599609375,
|
||||
132.1678924560547
|
||||
],
|
||||
"size": [
|
||||
400,
|
||||
252
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"links": [
|
||||
2
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "text_processed",
|
||||
"type": "STRING",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"cnr_id": "ComfyUI-Geeky-Kokoro-TTS",
|
||||
"ver": "443f4b0e1c503b47737b18c89f5067b9a6187038",
|
||||
"Node name for S&R": "GeekyKokoroTTS"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Welcome to ComfyUI",
|
||||
"🇺🇸 🚺 Heart ❤️",
|
||||
1,
|
||||
true,
|
||||
false,
|
||||
"🇺🇸 🚺 Sarah",
|
||||
0.5
|
||||
]
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
1,
|
||||
2,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
2,
|
||||
1,
|
||||
0,
|
||||
3,
|
||||
1,
|
||||
"AUDIO"
|
||||
],
|
||||
[
|
||||
3,
|
||||
3,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
4,
|
||||
3,
|
||||
1,
|
||||
4,
|
||||
1,
|
||||
"AUDIO"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.9646149645000188,
|
||||
"offset": [
|
||||
2147.454892156228,
|
||||
60.38347479123573
|
||||
]
|
||||
},
|
||||
"VHS_latentpreview": false,
|
||||
"VHS_latentpreviewrate": 0,
|
||||
"VHS_MetadataImage": true,
|
||||
"VHS_KeepIntermediate": true
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user