This commit is contained in:
wzhouxiff
2023-12-25 20:35:31 +08:00
parent 9910729448
commit bb00d47a01
71 changed files with 13026 additions and 29 deletions
+6
View File
@@ -0,0 +1,6 @@
# main/evaluation
.vscode/
*.pyc
gradio_temp
*.pth
+201
View File
@@ -0,0 +1,201 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
+79 -29
View File
@@ -1,29 +1,33 @@
<div align="center">
<!-- <div align="center"> -->
<!-- <h1>AnimateZero</h1> -->
<h3><b>MotionCtrl</b>: A Unified and Flexible
<!-- <h3><b>MotionCtrl</b>: A Unified and Flexible
Motion Controller
for Video Generation</h3>
for Video Generation</h3> -->
<!-- [![ Paper](https://img.shields.io/badge/Paper-MotionCtrl-red
)](https://wzhouxiff.github.io/projects/MotionCtrl/assets/paper/MotionCtrl.pdf) &ensp; [![ arXiv](https://img.shields.io/badge/arXiv-2312.03641-red
)](https://arxiv.org/pdf/2312.03641.pdf) &ensp; [![Porject Page](https://img.shields.io/badge/Project%20%20Page-MotionCtrl-red)
](https://wzhouxiff.github.io/projects/MotionCtrl/) &ensp; [![Demo](https://img.shields.io/badge/Demo-MotionCtrl-orange
)]() -->
<!-- [![ Paper](https://img.shields.io/badge/Paper-gray
)](https://wzhouxiff.github.io/projects/MotionCtrl/assets/paper/MotionCtrl.pdf) &ensp; [![ arXiv](https://img.shields.io/badge/arXiv-red
)](https://arxiv.org/pdf/2312.03641.pdf) &ensp; [![Porject Page](https://img.shields.io/badge/Project%20Page-green
)
](https://wzhouxiff.github.io/projects/MotionCtrl/) &ensp; [![Demo](https://img.shields.io/badge/Gradio%20Demo-orange
)](https://huggingface.co/spaces/TencentARC/MotionCtrl)
[Zhouxia Wang](https://vvictoryuki.github.io/website/)<sup>1,2</sup>, [Ziyang Yuan](https://github.com/jiangyzy)<sup>1,4</sup>, [Xintao Wang](https://xinntao.github.io/)<sup>1,3</sup>, [Tianshui Chen](http://tianshuichen.com/)<sup>6</sup>, [Menghan Xia](https://menghanxia.github.io/)<sup>3</sup>, [Ping Luo](http://luoping.me/)<sup>2,5</sup>, [Ying Shan](https://scholar.google.com/citations?hl=zh-CN&user=4oXBp9UAAAAJ)<sup>1,3</sup>
<sup>1</sup> ARC Lab, Tencent PCG, <sup>2</sup> The University of Hong Kong, <sup>3</sup> Tencent AI Lab, <sup>4</sup> Tsinghua University, <sup>5</sup> Shanghai AI Laboratory, <sup>6</sup> Guangdong University of Technology
[![ Paper](https://img.shields.io/badge/Paper-MotionCtrl-red
)](https://wzhouxiff.github.io/projects/MotionCtrl/assets/paper/MotionCtrl.pdf) &ensp; [![ arXiv](https://img.shields.io/badge/arXiv-2312.03641-red
)](https://arxiv.org/pdf/2312.03641.pdf) &ensp; [![Porject Page](https://img.shields.io/badge/Project%20%20Page-MotionCtrl-red)
](https://wzhouxiff.github.io/projects/MotionCtrl/) &ensp; [![Demo](https://img.shields.io/badge/Demo-MotionCtrl-orange
)]()
</div>
</div> -->
<!-- ## Results of MotionCtrl -->
Our proposed **MotionCtrl** is capable of independently controlling the complex camera motion and object motion of the generated videos, with **only a unified** model.
There are some results attained with **MotionCtrl** and more results are showcased in our [Project Page](https://wzhouxiff.github.io/projects/MotionCtrl/).
<!--
</br>
<!-- </br>
<video poster="" id="steve" autoplay controls muted loop playsinline height="100%" width="100%">
<source src="https://wzhouxiff.github.io/projects/MotionCtrl/assets/videos/teasers/camera_d971457c81bca597.mp4" type="video/mp4">
</video>
@@ -35,24 +39,64 @@ There are some results attained with **MotionCtrl** and more results are showcas
</video>
<video poster="" id="steve" autoplay controls muted loop playsinline height="100%" width="100%">
<source src="https://wzhouxiff.github.io/projects/MotionCtrl/assets/videos/teasers/s_curve_3_v1.mp4" type="video/mp4">
</video>
-->
https://github.com/TencentARC/MotionCtrl/assets/19488619/28a42fe7-e6df-49ec-b3ff-61c197b21819
https://github.com/TencentARC/MotionCtrl/assets/19488619/aa12d150-d49f-4415-aaf1-6e2e2c2fdbe4
https://github.com/TencentARC/MotionCtrl/assets/19488619/89feeac4-c152-4bb8-a8df-cdb2dcd74ccb
https://github.com/TencentARC/MotionCtrl/assets/19488619/e357ee60-8915-4cf0-a9d4-17fe32288d09
</video> -->
## Updating
- [ ] Code Release
- [ ] Gradio Demo Available
## 📝 Changelog
## Citation
- [x] 20231225: Release MotionCtrl depolyed on *LVDM/VideoCrafter*
- [x] 20231225: Gradio Demo Available. [![ Demo](https://img.shields.io/badge/Gradio%20Demo-orange
)](https://huggingface.co/spaces/TencentARC/MotionCtrl)
---
# MotionCtrl: A Unified and Flexible Motion Controller for Video Generation
[![ Paper](https://img.shields.io/badge/Paper-gray
)](https://wzhouxiff.github.io/projects/MotionCtrl/assets/paper/MotionCtrl.pdf) &ensp; [![ arXiv](https://img.shields.io/badge/arXiv-red
)](https://arxiv.org/pdf/2312.03641.pdf) &ensp; [![Porject Page](https://img.shields.io/badge/Project%20Page-green
)
](https://wzhouxiff.github.io/projects/MotionCtrl/) &ensp; [![ Demo](https://img.shields.io/badge/Gradio%20Demo-orange
)](https://huggingface.co/spaces/TencentARC/MotionCtrl)
---
🔥🔥 This is an official implement of [MotionCtrl: A Unified and Flexible Motion Controller for Video Generation](https://arxiv.org/pdf/2312.03641.pdf), which is capable of independently controlling the **complex camera motion** and **object motion** of the generated videos, with **only a unified** model.
There are some results attained with <b>MotionCtrl</b> and more results are showcased in our [Project Page](https://wzhouxiff.github.io/projects/MotionCtrl/).
<div align="center">
<img src="assets/hpxvu-3d8ym.gif", width="600">
<img src="assets/w3nb7-9vz5t.gif", width="600">
<img src="assets/62n2a-wuvsw.gif", width="600">
<img src="assets/ilw96-ak827.gif", width="600">
</div>
---
## ⚙️ Environment
conda create -n motionctrl python=3.10.6
conda activate motionctrl
pip install -r requirements.txt
## :running: Inference
1. Download the weights of MotionCtrl [motionctrl.pth](https://huggingface.co/TencentARC/MotionCtrl/blob/main/motionctrl.pth) and put it to `./checkpoints`.
2. Go into `configs/inference/run.sh` and set `condtype` as 'camera_motion', 'object_motion', or 'both'.
- `condtype=camera_motion` means only control the **camera motion** in the generated video.
- `condtype=object_motion` means only control the **object motion** in the generated video.
- `condtype=both` means control the camera motion and object motion in the generated video **simultaneously**.
1. Running scripts:
sh configs/inference/run.sh
## :books: Citation
If you make use of our work, please cite our paper.
```bibtex
@inproceedings{wang2023motionctrl,
@@ -62,3 +106,9 @@ If you make use of our work, please cite our paper.
year={2023}
}
```
## 🤗 Acknowledgment
The current version of **MotionCtrl** is built on [VideoCrafter](https://github.com/AILab-CVC/VideoCrafter). We appreciate the authors for sharing their awesome codebase.
## ❓ Contact
For any question, feel free to email `wzhoux@connect.hku.hk` or `zhouzi1212@gmail.com`.
Binary file not shown.

After

Width:  |  Height:  |  Size: 2.2 MiB

Binary file not shown.
Binary file not shown.
Binary file not shown.

After

Width:  |  Height:  |  Size: 2.3 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.5 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 849 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 945 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 12 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.3 MiB

+104
View File
@@ -0,0 +1,104 @@
model:
base_learning_rate: 0.0001
scale_lr: false
target: motionctrl.motionctrl.MotionCtrl
params:
# param for object motion control
omcm_config:
pretrained: ~
target: lvdm.modules.encoders.adapter.Adapter
params:
channels:
- 320
- 640
- 1280
- 1280
nums_rb: 2
cin: 128
sk: true
use_conv: false
linear_start: 0.00085
linear_end: 0.012
num_timesteps_cond: 1
log_every_t: 200
timesteps: 1000
first_stage_key: video
cond_stage_key: caption
cond_stage_trainable: false
conditioning_key: crossattn
image_size:
- 32
- 32
channels: 4
scale_by_std: false
scale_factor: 0.18215
use_ema: false
uncond_prob: 0.1
uncond_type: empty_seq
empty_params_only: true
scheduler_config:
target: utils.lr_scheduler.LambdaLRScheduler
interval: step
frequency: 100
params:
start_step: 0
final_decay_ratio: 0.01
decay_steps: 20000
unet_config:
target: lvdm.modules.networks.openaimodel3d_next.UNetModel
params:
in_channels: 4
out_channels: 4
model_channels: 320
attention_resolutions:
- 4
- 2
- 1
num_res_blocks: 2
channel_mult:
- 1
- 2
- 4
- 4
num_head_channels: 64
transformer_depth: 1
context_dim: 1024
use_linear: true
use_checkpoint: true
temporal_conv: true
temporal_attention: true
temporal_selfatt_only: true
use_relative_position: false
use_causal_attention: false
temporal_length: 16
use_image_dataset: false
addition_attention: true
first_stage_config:
target: lvdm.models.autoencoder.AutoencoderKL
params:
embed_dim: 4
monitor: val/rec_loss
ddconfig:
double_z: true
z_channels: 4
resolution: 256
in_channels: 3
out_ch: 3
ch: 128
ch_mult:
- 1
- 2
- 4
- 4
num_res_blocks: 2
attn_resolutions: []
dropout: 0.0
lossconfig:
target: torch.nn.Identity
cond_stage_config:
target: lvdm.modules.encoders.condition2.FrozenOpenCLIPEmbedder
params:
freeze: true
layer: penultimate
+52
View File
@@ -0,0 +1,52 @@
config="configs/inference/config_both.yaml"
ckpt='./checkpoints/motionctrl.pth'
condtype='both'
condtype='object_motion'
condtype='camera_motion'
cond_dir="examples/"
res_dir="./outputs/"
if [ ! -d $res_dir ]; then
mkdir -p $res_dir
fi
save_dir=$res_dir/$condtype'_seed'$seed
use_ddp=0
if [ $use_ddp == 0 ]; then
CUDA_VISIBLE_DEVICES=7 python 'main/evaluation/motionctrl_inference.py' \
--seed 1234 \
--ckpt_path $ckpt \
--base $config \
--savedir $save_dir \
--n_samples 5 \
--bs 1 --height 256 --width 256 \
--unconditional_guidance_scale 7.5 \
--ddim_steps 50 \
--ddim_eta 1.0 \
--condtype $condtype \
--cond_dir $cond_dir \
# --save_imgs
fi
if [ $use_ddp == 1 ]; then
python3 -m torch.distributed.launch \
--nproc_per_node=3 --nnodes=1 --master_port=23466 \
main/evaluation/ddp_wrapper.py \
--module 'inference' \
--seed 2000 \
--ckpt_path $ckpt \
--base $config \
--savedir $res_dir/$name \
--n_samples 3 \
--bs 1 --height 256 --width 256 \
--unconditional_guidance_scale 7.5 \
--ddim_steps 50 \
--ddim_eta 1.0 \
--condtype $condtype \
--cond_dir $cond_dir
fi
@@ -0,0 +1 @@
[[1.0, -4.493872218791495e-10, 5.58983348497577e-09, 1.9967236752904682e-09, -4.493872218791495e-10, 1.0, -6.144247333139674e-10, 1.0815730533408896e-09, 5.58983348497577e-09, -6.144247333139674e-10, 1.0, -7.984015226725205e-09], [0.9982863664627075, -0.0024742060340940952, 0.05846544727683067, -0.024547122418880463, 0.002410230925306678, 0.9999964237213135, 0.0011647245846688747, -0.003784072818234563, -0.05846811458468437, -0.0010218139505013824, 0.9982887506484985, -0.09103696048259735], [0.9933298230171204, -0.006303737405687571, 0.11513543128967285, -0.053876250982284546, 0.00586089538410306, 0.9999741315841675, 0.004184383898973465, -0.006566310301423073, -0.115158811211586, -0.0034816779661923647, 0.9933409690856934, -0.18525512516498566], [0.9849286675453186, -0.013619760051369667, 0.17242403328418732, -0.08322551101446152, 0.01256392989307642, 0.9998950958251953, 0.0072133541107177734, -0.004579910542815924, -0.17250417172908783, -0.004938316997140646, 0.9849964380264282, -0.28701746463775635], [0.9731453657150269, -0.022617166861891747, 0.2290775030851364, -0.11655563861131668, 0.02060025744140148, 0.9997251629829407, 0.011192308738827705, -0.0017426757840439677, -0.2292676568031311, -0.006172688212245703, 0.9733438491821289, -0.37736839056015015], [0.9582399725914001, -0.03294993191957474, 0.2840607464313507, -0.15743066370487213, 0.030182993039488792, 0.9994447827339172, 0.014113469049334526, -0.002769832033663988, -0.28436803817749023, -0.004950287751853466, 0.9587023854255676, -0.46959081292152405], [0.940129816532135, -0.03991429880261421, 0.3384712040424347, -0.22889098525047302, 0.03725311905145645, 0.9992027282714844, 0.01435780432075262, -0.0028311305213719606, -0.3387744128704071, -0.0008890923927538097, 0.9408671855926514, -0.5631460547447205], [0.9222924709320068, -0.044258520007133484, 0.38395029306411743, -0.2986142039299011, 0.04110203683376312, 0.9990199208259583, 0.01642671786248684, 0.0013055746676400304, -0.38430097699165344, 0.000630900904070586, 0.9232076406478882, -0.6414245367050171], [0.9061535000801086, -0.04851173609495163, 0.4201577305793762, -0.3483412563800812, 0.04521748423576355, 0.9988185167312622, 0.017803886905312538, 0.0010280977003276348, -0.4205249547958374, 0.0028654206544160843, 0.907276451587677, -0.7144853472709656], [0.8919307589530945, -0.05171844735741615, 0.4492044746875763, -0.37905213236808777, 0.04818608984351158, 0.9986518621444702, 0.019300933927297592, 0.00036871168413199484, -0.44959715008735657, 0.004430312197655439, 0.8932204246520996, -0.7976372241973877], [0.8792291879653931, -0.05425972864031792, 0.47329893708229065, -0.39671003818511963, 0.05076585337519646, 0.998507022857666, 0.02016463316977024, 0.001104982104152441, -0.4736863970756531, 0.00629808846861124, 0.8806710243225098, -0.8874085545539856], [0.8659296035766602, -0.0567130371928215, 0.49694016575813293, -0.4097800552845001, 0.05366959795355797, 0.9983500838279724, 0.020415671169757843, 0.0009228077251464128, -0.497278094291687, 0.008992047980427742, 0.8675445914268494, -0.9762357473373413], [0.8503361940383911, -0.055699657648801804, 0.5232837200164795, -0.44268566370010376, 0.054582174867391586, 0.9983546733856201, 0.01757136546075344, 0.005412018392235041, -0.5234014391899109, 0.013620397076010704, 0.8519773483276367, -1.069865107536316], [0.836037814617157, -0.05214058235287666, 0.5461887717247009, -0.4671085774898529, 0.05177384987473488, 0.9985294938087463, 0.01607322134077549, 0.008980141952633858, -0.5462236404418945, 0.014840473420917988, 0.8375079035758972, -1.1569048166275024], [0.82603919506073, -0.04987695440649986, 0.5614013671875, -0.4677649438381195, 0.05124447122216225, 0.9985973834991455, 0.013318539597094059, 0.012170637026429176, -0.5612781643867493, 0.017767081037163734, 0.8274364471435547, -1.2651430368423462], [0.8179472088813782, -0.0496118925511837, 0.573150098323822, -0.45822662115097046, 0.052784956991672516, 0.9985441565513611, 0.011104168370366096, 0.018991567194461823, -0.5728666186332703, 0.0211710836738348, 0.8193751573562622, -1.3895009756088257]]
@@ -0,0 +1 @@
[[0.9999999403953552, 3.8618797049139175e-10, -1.3441345814158012e-08, 1.3928219289027766e-07, 3.8618797049139175e-10, 1.0, -4.134579345560496e-10, -6.074658998045379e-09, -1.3441345814158012e-08, -4.134579345560496e-10, 1.0, 7.038884319854333e-08], [0.9994913339614868, 0.003077245783060789, -0.031741149723529816, 0.08338673412799835, -0.0030815028585493565, 0.999995231628418, -8.520588744431734e-05, 0.006532138213515282, 0.0317407064139843, 0.00018297435599379241, 0.9994961619377136, -0.02256060019135475], [0.9979938268661499, 0.0051255361177027225, -0.06310292333364487, 0.18344485759735107, -0.005117486696690321, 0.9999868869781494, 0.00028916727751493454, 0.018134046345949173, 0.06310353428125381, 3.434090831433423e-05, 0.9980069994926453, -0.030579563230276108], [0.9954646825790405, 0.00820203311741352, -0.0947771891951561, 0.29663264751434326, -0.00811922550201416, 0.9999662041664124, 0.0012593322899192572, 0.02404301054775715, 0.09478426724672318, -0.0004841022891923785, 0.9954977035522461, -0.02678978443145752], [0.9913660883903503, 0.012001598253846169, -0.13057230412960052, 0.4076530337333679, -0.011968829669058323, 0.999927818775177, 0.0010357286082580686, 0.024977533146739006, 0.1305752843618393, 0.0005360084469430149, 0.9914382100105286, -0.010779343545436859], [0.985666811466217, 0.017323914915323257, -0.16781197488307953, 0.509911060333252, -0.017399737611413002, 0.9998481273651123, 0.0010186078725382686, 0.023117201402783394, 0.16780413687229156, 0.0019158748909831047, 0.9858185052871704, 0.018053216859698296], [0.9784473180770874, 0.022585421800613403, -0.20525763928890228, 0.5957884192466736, -0.022850200533866882, 0.9997382760047913, 0.0010805513011291623, 0.020451901480555534, 0.2052282989025116, 0.003632916137576103, 0.9787073731422424, 0.03460140898823738], [0.9711515307426453, 0.026846906170248985, -0.23694702982902527, 0.6832671165466309, -0.02745947800576687, 0.999622642993927, 0.0007151798927225173, 0.012211678549647331, 0.23687675595283508, 0.005811895243823528, 0.971522331237793, 0.03236595541238785], [0.9641746878623962, 0.030338184908032417, -0.26352745294570923, 0.7764986157417297, -0.031404945999383926, 0.9995067715644836, 0.0001645474840188399, 0.0011497576488181949, 0.26340243220329285, 0.008117412216961384, 0.964651882648468, 0.022656364366412163], [0.9573631882667542, 0.0335896760225296, -0.2869274914264679, 0.8815275430679321, -0.03532479330897331, 0.9993755221366882, -0.0008711823611520231, -0.003618708113208413, 0.2867189943790436, 0.010969695635139942, 0.9579519033432007, 0.005283573176711798], [0.9507063627243042, 0.036557890474796295, -0.3079299330711365, 0.9931321740150452, -0.03846294432878494, 0.9992600679397583, -0.00011733790597645566, 0.0018704120302572846, 0.30769774317741394, 0.01195544097572565, 0.9514090418815613, -0.035360634326934814], [0.9448517560958862, 0.039408694952726364, -0.3251185715198517, 1.1025006771087646, -0.041503626853227615, 0.9991382360458374, 0.0004919985658489168, 0.007425118237733841, 0.32485777139663696, 0.013028733432292938, 0.9456731081008911, -0.09869624674320221], [0.940796971321106, 0.04081147164106369, -0.33650481700897217, 1.1961394548416138, -0.0429220013320446, 0.9990777373313904, 0.0011677180882543325, 0.019955899566411972, 0.336242139339447, 0.013344875536859035, 0.9416810274124146, -0.16835527122020721], [0.9376427531242371, 0.04111124947667122, -0.3451607823371887, 1.2392503023147583, -0.043144747614860535, 0.9990671873092651, 0.0017920633545145392, 0.03982722759246826, 0.34491249918937683, 0.013211555778980255, 0.938541829586029, -0.24618202447891235], [0.9353355765342712, 0.04122937470674515, -0.3513509929180145, 1.285768747329712, -0.043183211237192154, 0.9990646243095398, 0.0022769556380808353, 0.06841164082288742, 0.3511161506175995, 0.01304274145513773, 0.936241090297699, -0.3213619291782379], [0.9342393279075623, 0.041213057935237885, -0.3542574644088745, 1.3363462686538696, -0.04236872121691704, 0.9990919232368469, 0.0044970144517719746, 0.08925694227218628, 0.35412102937698364, 0.010808154009282589, 0.9351370930671692, -0.40201041102409363]]
@@ -0,0 +1 @@
[[1.0, 9.44418099280142e-10, 3.889182664806867e-08, 6.214055492392845e-09, 9.44418099280142e-10, 1.0, -1.0644604121756718e-11, -7.621465680784922e-10, 3.889182664806867e-08, -1.0644604121756718e-11, 1.0, -2.7145965475483536e-08], [0.9873979091644287, -0.007892023772001266, 0.15806053578853607, 0.4749181270599365, 0.008024877868592739, 0.9999678134918213, -0.00020230526570230722, 0.1585356593132019, -0.15805381536483765, 0.0014681711327284575, 0.9874294400215149, -0.2091633826494217], [0.9708925485610962, -0.011486345902085304, 0.23923994600772858, 0.8120080828666687, 0.012198254466056824, 0.9999244809150696, -0.0014952132478356361, 0.2486257702112198, -0.23920467495918274, 0.004370000213384628, 0.9709593057632446, -0.5957822799682617], [0.9619541168212891, -0.013188007287681103, 0.2728927433490753, 1.1486873626708984, 0.014017474837601185, 0.9999011754989624, -0.001090032048523426, 0.3114692270755768, -0.2728513777256012, 0.0048738280311226845, 0.962043821811676, -1.0323039293289185], [0.9586812257766724, -0.013936692848801613, 0.284140944480896, 1.5948307514190674, 0.014867136254906654, 0.9998888373374939, -0.0011181083973497152, 0.36000898480415344, -0.2840937674045563, 0.0052962712943553925, 0.958781898021698, -1.4377187490463257], [0.9583359360694885, -0.011928150430321693, 0.28539448976516724, 2.002793788909912, 0.014221147634088993, 0.9998810887336731, -0.0059633455239236355, 0.35464340448379517, -0.28528934717178345, 0.009773525409400463, 0.9583916068077087, -1.8953297138214111], [0.9584393501281738, -0.010862396098673344, 0.28508952260017395, 2.3351645469665527, 0.012857729569077492, 0.9999041557312012, -0.005128204356878996, 0.38934090733528137, -0.28500640392303467, 0.008580676279962063, 0.9584871530532837, -2.4214961528778076], [0.9587277173995972, -0.009760312736034393, 0.28415825963020325, 2.6858017444610596, 0.012186127714812756, 0.9999027848243713, -0.0067702098749578, 0.4173329174518585, -0.28406453132629395, 0.009953574277460575, 0.9587535262107849, -3.030754327774048], [0.9589635729789734, -0.0070899901911616325, 0.2834406793117523, 2.8917219638824463, 0.010482418350875378, 0.9998904466629028, -0.01045384630560875, 0.4001043438911438, -0.28333547711372375, 0.012996001169085503, 0.9589328169822693, -3.6960957050323486], [0.9590328931808472, -0.005921780597418547, 0.2832328677177429, 3.034579038619995, 0.00947173498570919, 0.9998928308486938, -0.01116593275219202, 0.4193899631500244, -0.2831363379955292, 0.013391205109655857, 0.9589861631393433, -4.384733200073242], [0.9593284726142883, -0.004661113955080509, 0.28225383162498474, 3.2042288780212402, 0.0076906089670956135, 0.9999240636825562, -0.009626304730772972, 0.4752484858036041, -0.28218749165534973, 0.011405492201447487, 0.9592914581298828, -5.098723411560059], [0.9591755867004395, -0.0035665074829012156, 0.2827887237071991, 3.263953924179077, 0.0062992447055876255, 0.9999418258666992, -0.008754877373576164, 0.4543868899345398, -0.282740980386734, 0.010178821161389351, 0.95914226770401, -5.7807512283325195], [0.9591742753982544, -0.003413048107177019, 0.2827949523925781, 3.3116462230682373, 0.003146615345031023, 0.9999940991401672, 0.0013963348465040326, 0.4299861788749695, -0.28279799222946167, -0.0004494813329074532, 0.9591793417930603, -6.478931903839111], [0.9585762619972229, -0.002857929328456521, 0.28482162952423096, 3.4120190143585205, -0.008201238699257374, 0.9992581605911255, 0.0376281812787056, 0.21596357226371765, -0.2847178280353546, -0.03840537369251251, 0.957841694355011, -7.178638935089111], [0.9572952389717102, -0.002719884505495429, 0.28909942507743835, 3.4662365913391113, -0.030634215101599693, 0.9933721423149109, 0.11078491061925888, -0.29767531156539917, -0.2874845862388611, -0.11491020023822784, 0.9508671164512634, -7.794575214385986], [0.9545961618423462, -0.005338957067579031, 0.29785510897636414, 3.5083835124969482, -0.06037351116538048, 0.9756243824958801, 0.2109788954257965, -1.0968165397644043, -0.29172107577323914, -0.21938219666481018, 0.9310050010681152, -8.306528091430664]]
+226
View File
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.2,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.28750000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.37500000000000006,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.4625000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.55,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.6375000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.7250000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.8125000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.9000000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-0.9875000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-1.0750000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-1.1625000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-1.2500000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-1.3375000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-1.4250000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
-1.5125000000000004,
0.0,
0.0,
1.0,
0.0
]
]
+226
View File
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.2
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.28750000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.37500000000000006
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.4625000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.55
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.6375000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.7250000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.8125000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.9000000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.9875000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.0750000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.1625000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.2500000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.3375000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.4250000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.5125000000000004
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.022500000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.045000000000000005
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.0675
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.09000000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.11250000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.135
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.15750000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.18000000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.2025
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.22500000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.24750000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.27
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.29250000000000004
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.31500000000000006
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.3375
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.045000000000000005
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.09000000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.135
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.18000000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.22500000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.27
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.31500000000000006
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.36000000000000004
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.405
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.45000000000000007
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.49500000000000005
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.54
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.5850000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.6300000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.675
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.1125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.225
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.3375
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.45
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.5625
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.675
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.7875
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.9
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.0125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.2375
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.35
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.4625000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.575
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.6875
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.225
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.45
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.675
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-0.9
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.35
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.575
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-1.8
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-2.025
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-2.25
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-2.475
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-2.7
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-2.9250000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-3.15
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
-3.375
]
]
+226
View File
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.2,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.28750000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.37500000000000006,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.4625000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.55,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.6375000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.7250000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.8125000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.9000000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.9875000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
1.0750000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
1.1625000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
1.2500000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
1.3375000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
1.4250000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
1.5125000000000004,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
]
]
+226
View File
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.2
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.28750000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.37500000000000006
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.4625000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.55
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.6375000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.7250000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.8125000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.9000000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.9875000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.0750000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.1625000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.2500000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.3375000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.4250000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.5125000000000004
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.022500000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.045000000000000005
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0675
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.09000000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.11250000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.135
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.15750000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.18000000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.2025
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.22500000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.24750000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.27
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.29250000000000004
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.31500000000000006
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.3375
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.045000000000000005
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.09000000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.135
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.18000000000000002
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.22500000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.27
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.31500000000000006
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.36000000000000004
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.405
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.45000000000000007
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.49500000000000005
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.54
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.5850000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.6300000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.675
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.1125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.225
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.3375
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.45
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.5625
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.675
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.7875
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.9
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.0125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.2375
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.35
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.4625000000000001
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.575
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.6875
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.225
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.45
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.675
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.9
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.125
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.35
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.575
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.8
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
2.025
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
2.25
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
2.475
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
2.7
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
2.9250000000000003
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
3.15
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
3.375
]
]
+226
View File
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
-0.2,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.28750000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.37500000000000006,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.4625000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.55,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.6375000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.7250000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.8125000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.9000000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-0.9875000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-1.0750000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-1.1625000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-1.2500000000000002,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-1.3375000000000001,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-1.4250000000000003,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
-1.5125000000000004,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.8
],
[
0.9914448613738104,
0.0,
0.13052619222005157,
0.13052619222005157,
0.0,
1.0,
0.0,
0.0,
-0.13052619222005157,
0.0,
0.9914448613738104,
1.7914448613738103
],
[
0.9659258262890683,
0.0,
0.25881904510252074,
0.25881904510252074,
0.0,
1.0,
0.0,
0.0,
-0.25881904510252074,
0.0,
0.9659258262890683,
1.7659258262890685
],
[
0.9238795325112867,
0.0,
0.3826834323650898,
0.3826834323650898,
0.0,
1.0,
0.0,
0.0,
-0.3826834323650898,
0.0,
0.9238795325112867,
1.7238795325112868
],
[
0.8660254037844387,
0.0,
0.49999999999999994,
0.49999999999999994,
0.0,
1.0,
0.0,
0.0,
-0.49999999999999994,
0.0,
0.8660254037844387,
1.6660254037844386
],
[
0.7933533402912353,
0.0,
0.6087614290087205,
0.6087614290087205,
0.0,
1.0,
0.0,
0.0,
-0.6087614290087205,
0.0,
0.7933533402912353,
1.5933533402912352
],
[
0.7071067811865476,
0.0,
0.7071067811865476,
0.7071067811865476,
0.0,
1.0,
0.0,
0.0,
-0.7071067811865476,
0.0,
0.7071067811865476,
1.5071067811865477
],
[
0.6087614290087207,
0.0,
0.7933533402912352,
0.7933533402912352,
0.0,
1.0,
0.0,
0.0,
-0.7933533402912352,
0.0,
0.6087614290087207,
1.4087614290087207
],
[
0.5000000000000001,
0.0,
0.8660254037844386,
0.8660254037844386,
0.0,
1.0,
0.0,
0.0,
-0.8660254037844386,
0.0,
0.5000000000000001,
1.3000000000000003
],
[
0.38268343236508984,
0.0,
0.9238795325112867,
0.9238795325112867,
0.0,
1.0,
0.0,
0.0,
-0.9238795325112867,
0.0,
0.38268343236508984,
1.1826834323650899
],
[
0.25881904510252096,
0.0,
0.9659258262890682,
0.9659258262890682,
0.0,
1.0,
0.0,
0.0,
-0.9659258262890682,
0.0,
0.25881904510252096,
1.058819045102521
],
[
0.1305261922200517,
0.0,
0.9914448613738104,
0.9914448613738104,
0.0,
1.0,
0.0,
0.0,
-0.9914448613738104,
0.0,
0.1305261922200517,
0.9305261922200517
],
[
6.123233995736766e-17,
0.0,
1.0,
1.0,
0.0,
1.0,
0.0,
0.0,
-1.0,
0.0,
6.123233995736766e-17,
0.8000000000000002
],
[
-0.13052619222005138,
0.0,
0.9914448613738105,
0.9914448613738105,
0.0,
1.0,
0.0,
0.0,
-0.9914448613738105,
0.0,
-0.13052619222005138,
0.6694738077799487
],
[
-0.25881904510252063,
0.0,
0.9659258262890683,
0.9659258262890683,
0.0,
1.0,
0.0,
0.0,
-0.9659258262890683,
0.0,
-0.25881904510252063,
0.5411809548974794
],
[
-0.3826834323650895,
0.0,
0.9238795325112868,
0.9238795325112868,
0.0,
1.0,
0.0,
0.0,
-0.9238795325112868,
0.0,
-0.3826834323650895,
0.41731656763491054
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.7000000000000002
],
[
0.9807852804032304,
0.0,
0.19509032201612825,
0.17558128981451543,
0.0,
1.0,
0.0,
0.0,
-0.19509032201612825,
0.0,
0.9807852804032304,
1.6827067523629076
],
[
0.9238795325112867,
0.0,
0.3826834323650898,
0.3444150891285808,
0.0,
1.0,
0.0,
0.0,
-0.3826834323650898,
0.0,
0.9238795325112867,
1.631491579260158
],
[
0.8314696123025452,
0.0,
0.5555702330196022,
0.500013209717642,
0.0,
1.0,
0.0,
0.0,
-0.5555702330196022,
0.0,
0.8314696123025452,
1.5483226510722907
],
[
0.7071067811865476,
0.0,
0.7071067811865476,
0.6363961030678928,
0.0,
1.0,
0.0,
0.0,
-0.7071067811865476,
0.0,
0.7071067811865476,
1.436396103067893
],
[
0.5555702330196023,
0.0,
0.8314696123025452,
0.7483226510722907,
0.0,
1.0,
0.0,
0.0,
-0.8314696123025452,
0.0,
0.5555702330196023,
1.3000132097176422
],
[
0.38268343236508984,
0.0,
0.9238795325112867,
0.831491579260158,
0.0,
1.0,
0.0,
0.0,
-0.9238795325112867,
0.0,
0.38268343236508984,
1.144415089128581
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.8827067523629074,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.9755812898145155
],
[
6.123233995736766e-17,
0.0,
1.0,
0.9,
0.0,
1.0,
0.0,
0.0,
-1.0,
0.0,
6.123233995736766e-17,
0.8
],
[
-0.1950903220161282,
0.0,
0.9807852804032304,
0.8827067523629074,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
-0.1950903220161282,
0.6244187101854847
],
[
-0.3826834323650897,
0.0,
0.9238795325112867,
0.831491579260158,
0.0,
1.0,
0.0,
0.0,
-0.9238795325112867,
0.0,
-0.3826834323650897,
0.4555849108714193
],
[
-0.555570233019602,
0.0,
0.8314696123025453,
0.7483226510722908,
0.0,
1.0,
0.0,
0.0,
-0.8314696123025453,
0.0,
-0.555570233019602,
0.2999867902823583
],
[
-0.7071067811865475,
0.0,
0.7071067811865476,
0.6363961030678928,
0.0,
1.0,
0.0,
0.0,
-0.7071067811865476,
0.0,
-0.7071067811865475,
0.1636038969321073
],
[
-0.8314696123025453,
0.0,
0.5555702330196022,
0.500013209717642,
0.0,
1.0,
0.0,
0.0,
-0.5555702330196022,
0.0,
-0.8314696123025453,
0.051677348927709255
],
[
-0.9238795325112867,
0.0,
0.3826834323650899,
0.34441508912858093,
0.0,
1.0,
0.0,
0.0,
-0.3826834323650899,
0.0,
-0.9238795325112867,
-0.031491579260158
],
[
-0.9807852804032304,
0.0,
0.1950903220161286,
0.17558128981451576,
0.0,
1.0,
0.0,
0.0,
-0.1950903220161286,
0.0,
-0.9807852804032304,
-0.08270675236290737
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
1.7000000000000002
],
[
0.9951847266721969,
0.0,
0.0980171403295606,
0.08821542629660455,
0.0,
1.0,
0.0,
0.0,
-0.0980171403295606,
0.0,
0.9951847266721969,
1.6956662540049772
],
[
0.9807852804032304,
0.0,
0.19509032201612825,
0.17558128981451543,
0.0,
1.0,
0.0,
0.0,
-0.19509032201612825,
0.0,
0.9807852804032304,
1.6827067523629076
],
[
0.9569403357322088,
0.0,
0.29028467725446233,
0.2612562095290161,
0.0,
1.0,
0.0,
0.0,
-0.29028467725446233,
0.0,
0.9569403357322088,
1.661246302158988
],
[
0.9238795325112867,
0.0,
0.3826834323650898,
0.3444150891285808,
0.0,
1.0,
0.0,
0.0,
-0.3826834323650898,
0.0,
0.9238795325112867,
1.631491579260158
],
[
0.881921264348355,
0.0,
0.47139673682599764,
0.4242570631433979,
0.0,
1.0,
0.0,
0.0,
-0.47139673682599764,
0.0,
0.881921264348355,
1.5937291379135194
],
[
0.8314696123025452,
0.0,
0.5555702330196022,
0.500013209717642,
0.0,
1.0,
0.0,
0.0,
-0.5555702330196022,
0.0,
0.8314696123025452,
1.5483226510722907
],
[
0.773010453362737,
0.0,
0.6343932841636455,
0.5709539557472809,
0.0,
1.0,
0.0,
0.0,
-0.6343932841636455,
0.0,
0.773010453362737,
1.4957094080264635
],
[
0.7071067811865476,
0.0,
0.7071067811865476,
0.6363961030678928,
0.0,
1.0,
0.0,
0.0,
-0.7071067811865476,
0.0,
0.7071067811865476,
1.436396103067893
],
[
0.6343932841636455,
0.0,
0.7730104533627369,
0.6957094080264632,
0.0,
1.0,
0.0,
0.0,
-0.7730104533627369,
0.0,
0.6343932841636455,
1.370953955747281
],
[
0.5555702330196023,
0.0,
0.8314696123025452,
0.7483226510722907,
0.0,
1.0,
0.0,
0.0,
-0.8314696123025452,
0.0,
0.5555702330196023,
1.3000132097176422
],
[
0.4713967368259978,
0.0,
0.8819212643483549,
0.7937291379135195,
0.0,
1.0,
0.0,
0.0,
-0.8819212643483549,
0.0,
0.4713967368259978,
1.2242570631433982
],
[
0.38268343236508984,
0.0,
0.9238795325112867,
0.831491579260158,
0.0,
1.0,
0.0,
0.0,
-0.9238795325112867,
0.0,
0.38268343236508984,
1.144415089128581
],
[
0.29028467725446233,
0.0,
0.9569403357322089,
0.861246302158988,
0.0,
1.0,
0.0,
0.0,
-0.9569403357322089,
0.0,
0.29028467725446233,
1.0612562095290161
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.8827067523629074,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.9755812898145155
],
[
0.09801714032956077,
0.0,
0.9951847266721968,
0.8956662540049771,
0.0,
1.0,
0.0,
0.0,
-0.9951847266721968,
0.0,
0.09801714032956077,
0.8882154262966048
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.81
],
[
0.9807852804032304,
0.0,
0.19509032201612825,
0.0019509032201612826,
0.0,
1.0,
0.0,
0.0,
-0.19509032201612825,
0.0,
0.9807852804032304,
0.8098078528040323
],
[
0.9238795325112867,
0.0,
0.3826834323650898,
0.003826834323650898,
0.0,
1.0,
0.0,
0.0,
-0.3826834323650898,
0.0,
0.9238795325112867,
0.8092387953251129
],
[
0.8314696123025452,
0.0,
0.5555702330196022,
0.005555702330196022,
0.0,
1.0,
0.0,
0.0,
-0.5555702330196022,
0.0,
0.8314696123025452,
0.8083146961230255
],
[
0.7071067811865476,
0.0,
0.7071067811865476,
0.007071067811865476,
0.0,
1.0,
0.0,
0.0,
-0.7071067811865476,
0.0,
0.7071067811865476,
0.8070710678118656
],
[
0.5555702330196023,
0.0,
0.8314696123025452,
0.008314696123025453,
0.0,
1.0,
0.0,
0.0,
-0.8314696123025452,
0.0,
0.5555702330196023,
0.805555702330196
],
[
0.38268343236508984,
0.0,
0.9238795325112867,
0.009238795325112868,
0.0,
1.0,
0.0,
0.0,
-0.9238795325112867,
0.0,
0.38268343236508984,
0.803826834323651
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.8019509032201614
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.8019509032201614
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.7019509032201614
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.6019509032201614
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.5019509032201613
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.4019509032201613
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.3019509032201613
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.2019509032201613
],
[
0.19509032201612833,
0.0,
0.9807852804032304,
0.009807852804032305,
0.0,
1.0,
0.0,
0.0,
-0.9807852804032304,
0.0,
0.19509032201612833,
0.10195090322016129
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9978589232386035,
-0.06540312923014306,
0.0,
0.0,
0.06540312923014306,
0.9978589232386035,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9914448613738104,
-0.13052619222005157,
0.0,
0.0,
0.13052619222005157,
0.9914448613738104,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9807852804032304,
-0.19509032201612825,
0.0,
0.0,
0.19509032201612825,
0.9807852804032304,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9659258262890683,
-0.25881904510252074,
0.0,
0.0,
0.25881904510252074,
0.9659258262890683,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9469301294951057,
-0.32143946530316153,
0.0,
0.0,
0.32143946530316153,
0.9469301294951057,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9238795325112867,
-0.3826834323650898,
0.0,
0.0,
0.3826834323650898,
0.9238795325112867,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.8968727415326884,
-0.44228869021900125,
0.0,
0.0,
0.44228869021900125,
0.8968727415326884,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.8660254037844387,
-0.49999999999999994,
0.0,
0.0,
0.49999999999999994,
0.8660254037844387,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.8314696123025452,
-0.5555702330196022,
0.0,
0.0,
0.5555702330196022,
0.8314696123025452,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.7933533402912353,
-0.6087614290087205,
0.0,
0.0,
0.6087614290087205,
0.7933533402912353,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.7518398074789774,
-0.6593458151000688,
0.0,
0.0,
0.6593458151000688,
0.7518398074789774,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.7071067811865476,
-0.7071067811865476,
0.0,
0.0,
0.7071067811865476,
0.7071067811865476,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.659345815100069,
-0.7518398074789773,
0.0,
0.0,
0.7518398074789773,
0.659345815100069,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.6087614290087207,
-0.7933533402912352,
0.0,
0.0,
0.7933533402912352,
0.6087614290087207,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.5555702330196024,
-0.8314696123025451,
0.0,
0.0,
0.8314696123025451,
0.5555702330196024,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
]
]
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9978589232386035,
0.06540312923014306,
0.0,
0.0,
-0.06540312923014306,
0.9978589232386035,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9914448613738104,
0.13052619222005157,
0.0,
0.0,
-0.13052619222005157,
0.9914448613738104,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9807852804032304,
0.19509032201612825,
0.0,
0.0,
-0.19509032201612825,
0.9807852804032304,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9659258262890683,
0.25881904510252074,
0.0,
0.0,
-0.25881904510252074,
0.9659258262890683,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9469301294951057,
0.32143946530316153,
0.0,
0.0,
-0.32143946530316153,
0.9469301294951057,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.9238795325112867,
0.3826834323650898,
0.0,
0.0,
-0.3826834323650898,
0.9238795325112867,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.8968727415326884,
0.44228869021900125,
0.0,
0.0,
-0.44228869021900125,
0.8968727415326884,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.8660254037844387,
0.49999999999999994,
0.0,
0.0,
-0.49999999999999994,
0.8660254037844387,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.8314696123025452,
0.5555702330196022,
0.0,
0.0,
-0.5555702330196022,
0.8314696123025452,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.7933533402912353,
0.6087614290087205,
0.0,
0.0,
-0.6087614290087205,
0.7933533402912353,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.7518398074789774,
0.6593458151000688,
0.0,
0.0,
-0.6593458151000688,
0.7518398074789774,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.7071067811865476,
0.7071067811865476,
0.0,
0.0,
-0.7071067811865476,
0.7071067811865476,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.659345815100069,
0.7518398074789773,
0.0,
0.0,
-0.7518398074789773,
0.659345815100069,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.6087614290087207,
0.7933533402912352,
0.0,
0.0,
-0.7933533402912352,
0.6087614290087207,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
],
[
0.5555702330196024,
0.8314696123025451,
0.0,
0.0,
-0.8314696123025451,
0.5555702330196024,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0
]
]
+226
View File
@@ -0,0 +1,226 @@
[
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.2,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.28750000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.37500000000000006,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.4625000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.55,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.6375000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.7250000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.8125000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.9000000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
0.9875000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
1.0750000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
1.1625000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
1.2500000000000002,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
1.3375000000000001,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
1.4250000000000003,
0.0,
0.0,
1.0,
0.0
],
[
1.0,
0.0,
0.0,
0.0,
0.0,
1.0,
0.0,
1.5125000000000004,
0.0,
0.0,
1.0,
0.0
]
]
@@ -0,0 +1 @@
[[0.9999999403953552, -3.2563843288535566e-10, 8.624932434919685e-10, -5.431840754965833e-09, -3.2563843288535566e-10, 1.0, -7.078895802870022e-10, -2.678919752696629e-09, 8.624932434919685e-10, -7.078895802870022e-10, 1.0, 7.303774474110014e-09], [0.9999998807907104, 0.00026603759033605456, -0.0003414043167140335, -0.030107486993074417, -0.0002659278397914022, 0.9999999403953552, 0.00032293255208060145, 0.011978899128735065, 0.0003414914826862514, -0.0003228419227525592, 0.9999999403953552, -0.07893969118595123], [0.9999994039535522, -0.0006866907933726907, -0.0008186722989194095, -0.050785429775714874, 0.0006877407431602478, 0.9999989867210388, 0.0012827541213482618, 0.023323407396674156, 0.0008177922572940588, -0.0012833180371671915, 0.9999988675117493, -0.1486724615097046], [0.999999463558197, -0.0005364732351154089, -0.0008647883078083396, -0.0673225075006485, 0.0005379383219406009, 0.9999984502792358, 0.0016958877677097917, 0.03169107437133789, 0.0008638793369755149, -0.0016963522648438811, 0.9999982118606567, -0.21477271616458893], [0.9999995231628418, -0.0005853328038938344, 0.000731398060452193, -0.08539554476737976, 0.0005843221442773938, 0.9999988675117493, 0.0013807439245283604, 0.037947509437799454, -0.0007322035962715745, -0.0013803173787891865, 0.9999987483024597, -0.2858566641807556], [0.9999935030937195, -0.0007727851625531912, 0.00352613371796906, -0.10591986775398254, 0.0007684561423957348, 0.9999989867210388, 0.0012290359009057283, 0.0426139310002327, -0.003527079476043582, -0.0012263187672942877, 0.999993085861206, -0.3645589053630829], [0.9999805688858032, -0.0012497249990701675, 0.006103655323386192, -0.12257411330938339, 0.0012419418198987842, 0.9999984502792358, 0.001278668874874711, 0.05143251642584801, -0.006105243694037199, -0.0012710647424682975, 0.9999805688858032, -0.44212815165519714], [0.9999697804450989, -0.001364423311315477, 0.0076533216051757336, -0.13916294276714325, 0.0013530527940019965, 0.9999980330467224, 0.0014907552395015955, 0.06076823174953461, -0.0076553407125175, -0.0014803558588027954, 0.9999696016311646, -0.5131306648254395], [0.9999656081199646, -0.001309191924519837, 0.00818517804145813, -0.1577269434928894, 0.0012902054004371166, 0.9999964833259583, 0.0023244714830070734, 0.07221967726945877, -0.008188189007341862, -0.0023138325195759535, 0.9999638199806213, -0.5706046223640442], [0.9999715089797974, -0.000632039678748697, 0.0075194560922682285, -0.1842648684978485, 0.0006101227481849492, 0.9999955892562866, 0.002916603581979871, 0.0780157595872879, -0.0075212628580629826, -0.0029119334649294615, 0.9999675154685974, -0.6207911372184753], [0.999971330165863, 9.749359742272645e-05, 0.007570990361273289, -0.2200128734111786, -0.00012177121971035376, 0.9999948740005493, 0.0032062893733382225, 0.07083319127559662, -0.007570638321340084, -0.003207120578736067, 0.9999662041664124, -0.671477735042572], [0.9999842047691345, 0.0011085873702540994, 0.005505275446921587, -0.265299528837204, -0.0011242963373661041, 0.9999953508377075, 0.002851187251508236, 0.06430258601903915, -0.00550208892673254, -0.002857332583516836, 0.9999808073043823, -0.7294802069664001], [0.9999891519546509, 0.0030521126464009285, -0.003507567336782813, -0.3109421133995056, -0.0030410285107791424, 0.9999904632568359, 0.003161099273711443, 0.06503432989120483, 0.003517185803502798, -0.0031503995414823294, 0.999988853931427, -0.7917969226837158], [0.9995473623275757, 0.00647346256300807, -0.029379382729530334, -0.3455933630466461, -0.006376372650265694, 0.9999739527702332, 0.0033971993252635, 0.06981948018074036, 0.02940060943365097, -0.003208327107131481, 0.9995625615119934, -0.8474093675613403], [0.9966378808021545, 0.011493867263197899, -0.08112204819917679, -0.3555213510990143, -0.011167873628437519, 0.9999276399612427, 0.004471173509955406, 0.0734858587384224, 0.08116757124662399, -0.00355018163099885, 0.9966941475868225, -0.9062771201133728], [0.9889613389968872, 0.01762101612985134, -0.14712226390838623, -0.33937495946884155, -0.01693679392337799, 0.999839186668396, 0.0059022014029324055, 0.0776127353310585, 0.14720259606838226, -0.0033452697098255157, 0.9891006946563721, -0.9548948407173157]]
@@ -0,0 +1 @@
[[1.0, -1.4532828274127496e-09, 3.928266045782891e-10, -8.700192566379883e-09, -1.4532828274127496e-09, 1.0, -1.7456260048565042e-10, 9.261455491405002e-11, 3.928266045782891e-10, -1.7456260048565042e-10, 1.0, -1.5881864712241622e-08], [0.9850640296936035, -0.04082970321178436, 0.16727784276008606, 0.20439539849758148, 0.03502606600522995, 0.9986825585365295, 0.03750050067901611, -0.15185177326202393, -0.16858860850334167, -0.03108130767941475, 0.9851963520050049, 0.10121102631092072], [0.9168016910552979, -0.09482394903898239, 0.387921541929245, 0.44365406036376953, 0.06679922342300415, 0.9941277503967285, 0.08513444662094116, -0.26796984672546387, -0.39371633529663086, -0.0521385557949543, 0.9177521467208862, 0.06753730028867722], [0.7643709182739258, -0.15802493691444397, 0.6251122951507568, 0.6479341983795166, 0.09576083719730377, 0.9865723252296448, 0.13230620324611664, -0.33551496267318726, -0.6376261115074158, -0.04126973822712898, 0.7692397236824036, -0.07632352411746979], [0.5624101758003235, -0.2065712958574295, 0.8006392121315002, 0.8095709085464478, 0.10458854585886002, 0.9782856106758118, 0.17893710732460022, -0.3449772298336029, -0.8202170729637146, -0.016898371279239655, 0.5718027949333191, -0.267223596572876], [0.3860713243484497, -0.23765023052692413, 0.8913311958312988, 1.017128586769104, 0.10842984914779663, 0.9712379574775696, 0.21198998391628265, -0.34939509630203247, -0.9160742163658142, 0.014803661964833736, 0.4007355272769928, -0.5046621561050415], [0.25270596146583557, -0.2560441493988037, 0.9330493807792664, 1.2611535787582397, 0.09987718611955643, 0.9661006331443787, 0.23806333541870117, -0.36809343099594116, -0.9623742699623108, 0.03303031623363495, 0.26971235871315, -0.8164812922477722], [0.12896282970905304, -0.26742202043533325, 0.9549105167388916, 1.4084848165512085, 0.09037449955940247, 0.9621138572692871, 0.25723403692245483, -0.3597045838832855, -0.9875227212905884, 0.05312592536211014, 0.14824506640434265, -1.1886308193206787], [-0.037818778306245804, -0.2752484083175659, 0.9606289863586426, 1.3755683898925781, 0.08146493136882782, 0.957267701625824, 0.277492493391037, -0.353816956281662, -0.9959584474563599, 0.08875200152397156, -0.013779602944850922, -1.6891316175460815], [-0.1970304548740387, -0.2752428948879242, 0.9409677982330322, 1.1786020994186401, 0.07280438393354416, 0.9530242681503296, 0.2940141260623932, -0.36850038170814514, -0.9776904582977295, 0.12643630802631378, -0.16773590445518494, -2.261430025100708], [-0.3677733540534973, -0.26668044924736023, 0.8908559679985046, 0.8089978098869324, 0.06243090331554413, 0.9487544894218445, 0.30978602170944214, -0.37012383341789246, -0.9278174042701721, 0.16954797506332397, -0.33227747678756714, -2.8139760494232178], [-0.5150132775306702, -0.2527574300765991, 0.8190696835517883, 0.4462648332118988, 0.050335872918367386, 0.9449706673622131, 0.32325950264930725, -0.381427526473999, -0.8557030558586121, 0.20771150290966034, -0.473949670791626, -3.2886314392089844], [-0.6352092623710632, -0.23657281696796417, 0.7352160215377808, 0.0826139897108078, 0.036419764161109924, 0.9416990280151367, 0.33447936177253723, -0.3719184398651123, -0.771480917930603, 0.23924078047275543, -0.5895600318908691, -3.677316904067993], [-0.7218965888023376, -0.21918001770973206, 0.6563730239868164, -0.2607730031013489, 0.024064550176262856, 0.9399895071983337, 0.340353786945343, -0.359468549489975, -0.6915825009346008, 0.26149556040763855, -0.6733006834983826, -3.9900803565979004], [-0.7831082344055176, -0.2019210308790207, 0.5881916880607605, -0.609000563621521, 0.015634668990969658, 0.9391284584999084, 0.3432103097438812, -0.35386255383491516, -0.621688961982727, 0.2779669761657715, -0.732282280921936, -4.316582202911377], [-0.8348898887634277, -0.1848600059747696, 0.5184455513954163, -0.9689992070198059, 0.007868430577218533, 0.93780916929245, 0.3470619320869446, -0.3402447998523712, -0.5503608584403992, 0.2938378155231476, -0.7815127968788147, -4.561704635620117]]
@@ -0,0 +1 @@
[[1.0, -1.9455232563858615e-11, -8.611562019034125e-10, -1.0473515388298438e-09, -1.9455232563858615e-11, 1.0, 8.674055917978762e-10, 1.8375034827045056e-09, -8.611562019034125e-10, 8.674055917978762e-10, 1.0, 8.45981773522908e-09], [0.9973401427268982, -0.0032562706619501114, 0.07281485944986343, -0.07197967171669006, 0.00329946493729949, 0.9999944567680359, -0.00047293686657212675, -0.0565447174012661, -0.07281292229890823, 0.0007119301008060575, 0.9973453879356384, -0.3188611567020416], [0.9906007051467896, -0.0066957552917301655, 0.13662146031856537, -0.15415722131729126, 0.006810691673308611, 0.9999766945838928, -0.00037385127507150173, -0.10014614462852478, -0.1366157978773117, 0.0013008243404328823, 0.9906232953071594, -0.6077051758766174], [0.9827210903167725, -0.009423106908798218, 0.18485280871391296, -0.23053960502147675, 0.009786692447960377, 0.9999515414237976, -0.0010545575059950352, -0.11272216588258743, -0.18483391404151917, 0.002845433307811618, 0.9827656745910645, -0.8897562623023987], [0.9756309986114502, -0.011318936944007874, 0.2191258817911148, -0.2929331362247467, 0.01199379749596119, 0.9999265670776367, -0.0017497432418167591, -0.10348402708768845, -0.21908999979496002, 0.004335256293416023, 0.9756950736045837, -1.1368639469146729], [0.9685831069946289, -0.012191502377390862, 0.24839124083518982, -0.379209041595459, 0.013217172585427761, 0.9999096393585205, -0.0024619612377136946, -0.12078691273927689, -0.24833877384662628, 0.005667644087225199, 0.9686566591262817, -1.3666095733642578], [0.9615083932876587, -0.012519675306975842, 0.2744903266429901, -0.5242342352867126, 0.013806014321744442, 0.9999009370803833, -0.0027547888457775116, -0.13015854358673096, -0.2744286358356476, 0.006438371259719133, 0.9615859389305115, -1.5762972831726074], [0.9528889656066895, -0.014083069749176502, 0.3029923439025879, -0.6702165007591248, 0.015383570455014706, 0.9998798370361328, -0.0019058430334553123, -0.14860114455223083, -0.3029291331768036, 0.00647716224193573, 0.9529911279678345, -1.7173497676849365], [0.9457509517669678, -0.015159872360527515, 0.3245387375354767, -0.8001917600631714, 0.016729505732655525, 0.9998579621315002, -0.0020466779824346304, -0.15521110594272614, -0.32446160912513733, 0.007365020923316479, 0.9458702206611633, -1.8571592569351196], [0.9417706727981567, -0.015540024265646935, 0.33589670062065125, -0.9244012236595154, 0.017394419759511948, 0.9998455047607422, -0.0025124624371528625, -0.1416105479001999, -0.3358057737350464, 0.008208892308175564, 0.9418954849243164, -1.9969842433929443], [0.9381415247917175, -0.016401933506131172, 0.34586337208747864, -1.0128529071807861, 0.01851990446448326, 0.9998245239257812, -0.0028197101783007383, -0.11503250896930695, -0.3457564413547516, 0.009050644934177399, 0.9382806420326233, -2.080850124359131], [0.9357938766479492, -0.016873901709914207, 0.3521435558795929, -1.1008079051971436, 0.01865064539015293, 0.9998247027397156, -0.0016533475136384368, -0.048780668526887894, -0.3520539104938507, 0.008114897646009922, 0.9359445571899414, -2.088500499725342], [0.9334627985954285, -0.016492612659931183, 0.3582947850227356, -1.2417148351669312, 0.017931735143065453, 0.9998390078544617, -0.0006939812446944416, 0.01625901833176613, -0.35822567343711853, 0.007072653621435165, 0.9336082339286804, -2.0250792503356934], [0.9281281232833862, -0.016213543713092804, 0.3719078302383423, -1.4365860223770142, 0.01746317930519581, 0.9998475313186646, 8.084020009846427e-06, 0.060201361775398254, -0.37185123562812805, 0.006487190257757902, 0.9282696843147278, -1.8963005542755127], [0.9220678806304932, -0.017285069450736046, 0.3866421580314636, -1.604330062866211, 0.01880805380642414, 0.9998230934143066, -0.00015593231364618987, 0.0799601748585701, -0.38657107949256897, 0.007415766827762127, 0.9222298264503479, -1.7388361692428589], [0.9185494184494019, -0.018356993794441223, 0.3948797583580017, -1.7530856132507324, 0.019847355782985687, 0.9998030066490173, 0.0003104731731582433, 0.10000917315483093, -0.3948076665401459, 0.007552135270088911, 0.9187328219413757, -1.5337872505187988]]
+59
View File
@@ -0,0 +1,59 @@
88, 173
89, 173
91, 173
92, 173
93, 173
93, 174
95, 175
97, 176
98, 176
100, 178
103, 180
104, 181
105, 181
106, 182
107, 182
108, 182
110, 184
111, 185
113, 186
115, 186
116, 187
116, 188
119, 189
122, 190
125, 191
128, 192
129, 192
130, 192
133, 193
134, 193
140, 194
146, 194
150, 194
151, 194
153, 194
155, 194
156, 194
157, 194
158, 194
159, 194
162, 193
165, 192
168, 191
169, 190
170, 190
172, 189
173, 188
173, 187
174, 187
175, 186
178, 185
179, 183
180, 183
181, 183
183, 181
185, 180
185, 179
185, 179
185, 179
+80
View File
@@ -0,0 +1,80 @@
206, 146
206, 147
205, 149
204, 151
203, 151
202, 154
201, 156
200, 159
199, 161
198, 163
197, 164
197, 165
197, 166
195, 168
193, 171
191, 173
189, 176
188, 177
187, 178
187, 179
186, 180
185, 181
184, 181
184, 182
182, 184
180, 186
179, 187
177, 189
175, 191
174, 191
173, 192
172, 193
171, 193
169, 195
167, 197
164, 198
161, 200
160, 201
158, 202
156, 204
155, 204
154, 204
153, 205
151, 206
150, 206
149, 206
147, 207
146, 207
143, 209
139, 210
138, 210
135, 211
135, 212
134, 212
133, 212
132, 212
131, 212
130, 212
129, 212
128, 212
127, 212
126, 213
124, 213
122, 213
120, 213
120, 214
119, 214
118, 214
116, 214
115, 214
114, 214
113, 214
112, 214
111, 214
110, 214
110, 215
109, 215
108, 215
108, 215
108, 215
+150
View File
@@ -0,0 +1,150 @@
73, 149
74, 149
76, 149
82, 149
83, 149
90, 149
98, 149
102, 149
104, 149
105, 149
106, 150
108, 152
111, 154
114, 157
116, 158
117, 160
118, 161
119, 164
121, 166
122, 168
122, 169
123, 171
123, 172
124, 174
125, 177
126, 179
126, 181
126, 183
126, 184
126, 185
127, 186
127, 187
127, 188
127, 189
127, 190
127, 191
128, 192
128, 193
129, 194
130, 196
130, 197
131, 198
131, 199
134, 201
135, 202
137, 203
138, 204
139, 204
139, 205
140, 206
142, 206
142, 207
147, 209
151, 211
153, 212
154, 212
156, 212
158, 213
159, 214
161, 214
162, 214
163, 214
164, 214
165, 214
167, 214
168, 214
169, 214
170, 214
171, 214
172, 213
173, 213
175, 212
176, 212
177, 212
178, 211
179, 211
179, 208
182, 205
184, 203
184, 202
186, 200
187, 199
187, 198
188, 196
190, 194
192, 191
193, 189
193, 188
194, 188
194, 187
195, 184
196, 183
197, 182
197, 180
198, 180
198, 179
199, 178
199, 177
200, 176
200, 175
201, 174
202, 173
203, 171
204, 170
204, 169
205, 169
205, 167
207, 165
208, 163
209, 163
209, 162
210, 162
211, 162
212, 162
213, 162
215, 162
216, 162
217, 163
219, 164
220, 164
221, 165
223, 166
224, 166
224, 167
225, 167
225, 168
226, 169
227, 171
228, 173
228, 174
229, 175
229, 176
229, 177
229, 180
229, 181
229, 182
229, 183
229, 184
229, 185
230, 186
230, 187
230, 187
230, 189
224, 192
223, 193
220, 194
219, 195
218, 195
218, 195
218, 195
+137
View File
@@ -0,0 +1,137 @@
176, 109
177, 110
177, 111
177, 112
177, 114
178, 118
179, 122
180, 125
180, 126
180, 127
180, 128
180, 129
180, 130
180, 131
180, 133
180, 134
179, 136
177, 138
177, 139
176, 140
176, 141
174, 143
172, 145
170, 147
170, 148
167, 150
163, 154
159, 158
158, 158
155, 160
154, 161
152, 163
151, 164
149, 165
147, 166
145, 167
143, 169
140, 171
137, 172
135, 174
133, 175
131, 176
130, 176
128, 177
127, 177
125, 178
123, 179
121, 179
119, 179
116, 179
114, 180
112, 180
111, 180
110, 180
108, 180
104, 181
99, 181
95, 181
94, 181
93, 179
92, 179
92, 176
91, 173
91, 171
91, 170
91, 168
91, 165
91, 164
91, 163
91, 162
91, 161
93, 159
94, 159
99, 156
100, 156
105, 153
106, 152
110, 149
112, 148
113, 148
114, 147
115, 146
117, 145
119, 144
120, 144
121, 143
122, 142
123, 142
124, 142
125, 141
126, 141
127, 140
128, 140
128, 139
129, 138
128, 138
126, 137
125, 136
123, 136
122, 136
121, 135
120, 135
119, 135
118, 135
117, 135
116, 135
115, 135
112, 134
111, 134
109, 134
107, 133
104, 133
103, 133
102, 133
100, 133
99, 133
96, 133
94, 133
93, 133
92, 133
91, 133
89, 133
88, 133
87, 133
86, 133
85, 133
84, 133
82, 133
81, 133
80, 133
79, 133
78, 133
77, 133
76, 133
76, 133
76, 133
76, 133
+48
View File
@@ -0,0 +1,48 @@
121, 124
123, 124
124, 124
125, 124
127, 124
130, 124
131, 124
135, 124
140, 124
143, 124
144, 124
145, 124
149, 124
152, 124
154, 124
155, 124
156, 124
159, 125
160, 125
163, 125
166, 126
168, 126
169, 126
170, 126
173, 126
176, 126
177, 126
178, 126
179, 126
180, 126
181, 126
183, 126
184, 126
185, 126
186, 126
187, 126
188, 126
189, 126
190, 126
191, 126
192, 126
193, 126
194, 126
195, 126
196, 126
196, 126
196, 126
196, 126
+205
View File
@@ -0,0 +1,205 @@
103, 89
104, 89
106, 89
107, 89
108, 89
109, 89
110, 89
111, 89
112, 89
113, 89
114, 89
115, 89
116, 89
117, 89
118, 89
119, 89
120, 89
122, 89
123, 89
124, 89
125, 89
126, 89
127, 88
128, 88
129, 88
130, 88
131, 88
133, 87
136, 86
137, 86
138, 86
139, 86
140, 86
141, 86
142, 86
143, 86
144, 86
145, 86
146, 87
147, 87
148, 87
149, 87
148, 87
146, 87
145, 88
144, 88
142, 89
141, 89
140, 90
140, 91
138, 91
137, 92
136, 92
136, 93
135, 93
134, 93
133, 93
132, 93
131, 93
130, 93
129, 93
128, 93
127, 92
125, 92
124, 92
123, 92
122, 92
121, 92
120, 92
119, 92
118, 92
117, 92
116, 92
115, 92
113, 92
112, 92
111, 92
110, 92
109, 92
108, 92
108, 91
108, 90
109, 90
110, 90
111, 89
112, 89
113, 89
114, 89
115, 89
115, 88
116, 88
117, 88
118, 88
118, 87
119, 87
120, 87
121, 87
122, 86
123, 86
124, 86
125, 86
126, 85
127, 85
128, 85
129, 85
130, 85
131, 85
132, 85
133, 85
134, 85
135, 85
136, 85
137, 85
138, 85
139, 85
140, 85
141, 85
142, 85
143, 85
143, 84
144, 84
145, 84
146, 84
147, 84
148, 84
149, 84
148, 84
147, 84
145, 84
144, 84
143, 84
142, 84
141, 84
140, 85
139, 85
138, 85
137, 86
136, 86
136, 87
135, 87
134, 87
133, 87
132, 88
131, 88
130, 88
129, 88
129, 89
128, 89
127, 89
126, 89
125, 89
124, 90
123, 90
122, 90
121, 90
120, 91
119, 91
118, 91
117, 91
116, 91
115, 91
114, 91
113, 91
112, 91
111, 91
110, 91
109, 91
109, 90
108, 90
110, 90
111, 90
113, 90
114, 90
115, 90
116, 90
118, 90
120, 90
121, 90
122, 90
123, 90
124, 90
126, 90
127, 90
128, 90
129, 90
130, 90
131, 90
132, 90
133, 90
134, 90
135, 90
136, 90
137, 90
138, 90
139, 90
140, 90
141, 89
142, 89
143, 89
144, 89
145, 89
146, 89
147, 89
147, 89
147, 89
+156
View File
@@ -0,0 +1,156 @@
125, 86
127, 86
129, 86
130, 86
131, 86
133, 86
134, 86
135, 86
137, 86
138, 86
139, 86
140, 86
141, 86
142, 86
141, 86
139, 86
139, 87
138, 87
137, 87
136, 87
135, 87
134, 87
134, 88
133, 88
132, 88
131, 88
130, 88
129, 88
128, 88
127, 88
128, 88
129, 88
130, 88
131, 88
132, 88
133, 88
134, 88
135, 88
136, 88
137, 88
138, 88
139, 88
140, 88
141, 88
140, 88
139, 88
138, 88
136, 88
134, 88
133, 88
131, 88
130, 88
129, 88
128, 88
127, 88
128, 88
129, 88
130, 88
132, 87
133, 87
134, 86
136, 86
137, 86
138, 85
139, 85
140, 85
140, 84
141, 84
139, 85
138, 85
137, 86
136, 86
135, 86
134, 86
133, 86
132, 86
131, 86
130, 86
129, 86
128, 86
127, 86
126, 86
127, 86
129, 86
130, 86
132, 86
133, 86
135, 86
137, 86
138, 86
139, 86
140, 86
141, 87
142, 87
143, 87
144, 88
143, 88
142, 88
141, 88
139, 88
138, 88
137, 88
136, 88
135, 88
134, 88
133, 88
131, 88
130, 88
129, 88
128, 88
127, 88
128, 88
129, 88
130, 88
132, 88
133, 88
134, 88
136, 88
137, 88
138, 88
139, 88
140, 88
141, 88
142, 88
143, 88
144, 88
143, 88
142, 88
141, 88
140, 88
139, 88
138, 88
137, 88
135, 89
135, 90
133, 90
131, 91
130, 91
129, 91
128, 91
127, 91
128, 91
129, 91
130, 91
132, 91
133, 90
134, 89
135, 89
136, 89
137, 89
138, 89
138, 88
139, 88
140, 88
140, 88
140, 88
+373
View File
@@ -0,0 +1,373 @@
117, 102
114, 102
109, 102
106, 102
105, 102
102, 102
99, 102
97, 102
96, 102
95, 102
93, 102
89, 102
85, 103
82, 103
81, 103
80, 103
79, 103
78, 103
76, 103
74, 104
73, 104
72, 104
71, 104
70, 105
69, 105
68, 105
67, 105
66, 106
64, 107
63, 108
62, 108
61, 108
61, 109
60, 109
59, 109
58, 109
57, 110
56, 110
55, 111
54, 111
53, 111
52, 111
52, 112
51, 112
50, 112
50, 113
49, 113
48, 113
46, 114
46, 115
45, 115
45, 116
44, 116
43, 117
42, 117
41, 117
41, 118
40, 118
41, 118
41, 119
42, 119
43, 119
44, 119
46, 119
47, 119
48, 119
49, 119
50, 119
51, 119
52, 119
53, 119
54, 119
55, 119
56, 118
58, 118
59, 118
61, 118
63, 118
64, 117
67, 117
70, 117
71, 117
73, 117
75, 116
76, 116
77, 116
80, 116
82, 116
83, 116
84, 116
85, 116
88, 116
91, 116
94, 116
97, 116
98, 116
100, 116
101, 117
102, 117
104, 117
105, 117
106, 117
107, 117
108, 117
109, 117
110, 117
111, 117
115, 117
119, 117
123, 117
124, 117
128, 117
129, 117
132, 117
134, 117
135, 117
136, 117
138, 117
139, 117
140, 117
141, 117
142, 116
145, 116
146, 116
148, 116
149, 116
151, 115
152, 115
153, 115
154, 115
155, 114
156, 114
157, 114
158, 114
159, 114
162, 114
163, 113
164, 113
165, 113
166, 113
167, 113
168, 113
169, 113
170, 113
171, 113
172, 113
173, 113
174, 113
175, 113
178, 113
181, 113
182, 113
183, 113
184, 113
185, 113
187, 113
188, 113
189, 113
191, 113
192, 113
193, 113
194, 113
195, 113
196, 113
197, 113
198, 113
199, 113
200, 113
201, 113
202, 113
203, 113
202, 113
201, 113
200, 113
198, 113
197, 113
196, 113
195, 112
194, 112
193, 112
192, 112
191, 111
190, 111
189, 111
188, 110
187, 110
186, 110
185, 110
184, 110
183, 110
182, 110
181, 110
180, 110
179, 110
178, 110
177, 110
175, 110
173, 110
172, 110
171, 110
170, 110
168, 110
167, 110
165, 110
164, 110
163, 110
161, 111
159, 111
155, 111
153, 111
151, 111
151, 112
150, 112
149, 112
148, 112
147, 112
145, 112
143, 113
142, 113
140, 113
139, 113
138, 113
136, 113
135, 113
134, 113
133, 114
131, 114
130, 114
128, 115
127, 115
126, 115
125, 115
124, 115
122, 115
121, 115
120, 115
118, 116
115, 116
113, 116
111, 116
109, 117
106, 117
103, 117
102, 117
100, 117
98, 117
97, 117
95, 117
94, 117
93, 117
92, 117
91, 117
90, 117
89, 117
88, 117
87, 117
86, 117
85, 117
84, 117
83, 117
84, 117
85, 117
87, 117
88, 117
89, 117
90, 117
92, 117
93, 117
95, 117
97, 117
99, 117
101, 117
103, 117
104, 117
105, 117
106, 117
107, 117
108, 117
109, 117
110, 117
112, 117
113, 117
114, 117
116, 117
117, 117
118, 117
119, 117
120, 117
121, 117
123, 117
124, 117
125, 117
126, 117
127, 117
129, 117
130, 117
131, 117
133, 117
134, 117
135, 117
136, 117
137, 117
138, 117
139, 117
140, 117
141, 117
142, 117
143, 117
145, 117
146, 117
147, 117
148, 117
149, 117
150, 117
149, 117
148, 117
147, 117
146, 117
144, 117
143, 118
142, 118
141, 118
140, 118
139, 118
138, 118
136, 118
135, 118
132, 119
131, 119
130, 119
129, 119
127, 119
126, 119
124, 119
123, 119
122, 119
121, 119
119, 119
118, 119
117, 119
115, 119
114, 119
113, 119
112, 119
111, 119
110, 119
109, 119
108, 119
107, 119
106, 119
107, 119
108, 119
109, 119
110, 119
112, 119
113, 119
114, 119
115, 119
116, 119
117, 119
118, 119
119, 119
120, 119
121, 119
122, 119
123, 119
124, 119
125, 119
126, 119
127, 119
127, 119
127, 119
127, 119
+106
View File
@@ -0,0 +1,106 @@
# adopted from
# https://github.com/openai/improved-diffusion/blob/main/improved_diffusion/gaussian_diffusion.py
# and
# https://github.com/lucidrains/denoising-diffusion-pytorch/blob/7706bdfc6f527f58d33f84b7b522e61e6e3164b3/denoising_diffusion_pytorch/denoising_diffusion_pytorch.py
# and
# https://github.com/openai/guided-diffusion/blob/0ba878e517b276c45d1195eb29f6f5f72659a05b/guided_diffusion/nn.py
#
# thanks!
import numpy as np
from einops import repeat
import torch
import torch.nn as nn
import torch.nn.functional as F
from utils.utils import instantiate_from_config
def disabled_train(self, mode=True):
"""Overwrite model.train with this function to make sure train/eval mode
does not change anymore."""
return self
def zero_module(module):
"""
Zero out the parameters of a module and return it.
"""
for p in module.parameters():
p.detach().zero_()
return module
def scale_module(module, scale):
"""
Scale the parameters of a module and return it.
"""
for p in module.parameters():
p.detach().mul_(scale)
return module
def conv_nd(dims, *args, **kwargs):
"""
Create a 1D, 2D, or 3D convolution module.
"""
if dims == 1:
return nn.Conv1d(*args, **kwargs)
elif dims == 2:
return nn.Conv2d(*args, **kwargs)
elif dims == 3:
return nn.Conv3d(*args, **kwargs)
raise ValueError(f"unsupported dimensions: {dims}")
def linear(*args, **kwargs):
"""
Create a linear module.
"""
return nn.Linear(*args, **kwargs)
def avg_pool_nd(dims, *args, **kwargs):
"""
Create a 1D, 2D, or 3D average pooling module.
"""
if dims == 1:
return nn.AvgPool1d(*args, **kwargs)
elif dims == 2:
return nn.AvgPool2d(*args, **kwargs)
elif dims == 3:
return nn.AvgPool3d(*args, **kwargs)
raise ValueError(f"unsupported dimensions: {dims}")
def nonlinearity(type='silu'):
if type == 'silu':
return nn.SiLU()
elif type == 'leaky_relu':
return nn.LeakyReLU()
class GroupNormSpecific(nn.GroupNorm):
def forward(self, x):
return super().forward(x.float()).type(x.dtype)
def normalization(channels, num_groups=32):
"""
Make a standard normalization layer.
:param channels: number of input channels.
:return: an nn.Module for normalization.
"""
return GroupNormSpecific(num_groups, channels)
class HybridConditioner(nn.Module):
def __init__(self, c_concat_config, c_crossattn_config):
super().__init__()
self.concat_conditioner = instantiate_from_config(c_concat_config)
self.crossattn_conditioner = instantiate_from_config(c_crossattn_config)
def forward(self, c_concat, c_crossattn):
c_concat = self.concat_conditioner(c_concat)
c_crossattn = self.crossattn_conditioner(c_crossattn)
return {'c_concat': [c_concat], 'c_crossattn': [c_crossattn]}
+142
View File
@@ -0,0 +1,142 @@
import os, math
import numpy as np
from inspect import isfunction
import torch
from torch import nn
import torch.nn.functional as F
import torch.distributed as dist
def gather_data(data, return_np=True):
''' gather data from multiple processes to one list '''
data_list = [torch.zeros_like(data) for _ in range(dist.get_world_size())]
dist.all_gather(data_list, data) # gather not supported with NCCL
if return_np:
data_list = [data.cpu().numpy() for data in data_list]
return data_list
def autocast(f):
def do_autocast(*args, **kwargs):
with torch.cuda.amp.autocast(enabled=True,
dtype=torch.get_autocast_gpu_dtype(),
cache_enabled=torch.is_autocast_cache_enabled()):
return f(*args, **kwargs)
return do_autocast
def extract_into_tensor(a, t, x_shape):
b, *_ = t.shape
out = a.gather(-1, t)
return out.reshape(b, *((1,) * (len(x_shape) - 1)))
def noise_like(shape, device, repeat=False):
repeat_noise = lambda: torch.randn((1, *shape[1:]), device=device).repeat(shape[0], *((1,) * (len(shape) - 1)))
noise = lambda: torch.randn(shape, device=device)
return repeat_noise() if repeat else noise()
def default(val, d):
if exists(val):
return val
return d() if isfunction(d) else d
def exists(val):
return val is not None
def identity(*args, **kwargs):
return nn.Identity()
def uniq(arr):
return{el: True for el in arr}.keys()
def mean_flat(tensor):
"""
Take the mean over all non-batch dimensions.
"""
return tensor.mean(dim=list(range(1, len(tensor.shape))))
def ismap(x):
if not isinstance(x, torch.Tensor):
return False
return (len(x.shape) == 4) and (x.shape[1] > 3)
def isimage(x):
if not isinstance(x,torch.Tensor):
return False
return (len(x.shape) == 4) and (x.shape[1] == 3 or x.shape[1] == 1)
def max_neg_value(t):
return -torch.finfo(t.dtype).max
def shape_to_str(x):
shape_str = "x".join([str(x) for x in x.shape])
return shape_str
def init_(tensor):
dim = tensor.shape[-1]
std = 1 / math.sqrt(dim)
tensor.uniform_(-std, std)
return tensor
#import deepspeed
#ckpt = deepspeed.checkpointing.checkpoint
ckpt = torch.utils.checkpoint.checkpoint
def checkpoint(func, inputs, params, flag):
"""
Evaluate a function without caching intermediate activations, allowing for
reduced memory at the expense of extra compute in the backward pass.
:param func: the function to evaluate.
:param inputs: the argument sequence to pass to `func`.
:param params: a sequence of parameters `func` depends on but does not
explicitly take as arguments.
:param flag: if False, disable gradient checkpointing.
"""
if flag:
try:
return ckpt(func, *inputs)
except:
args = tuple(inputs) + tuple(params)
return CheckpointFunction.apply(func, len(inputs), *args)
else:
return func(*inputs)
class CheckpointFunction(torch.autograd.Function):
@staticmethod
@torch.cuda.amp.custom_fwd
def forward(ctx, run_function, length, *args):
ctx.run_function = run_function
ctx.input_tensors = list(args[:length])
ctx.input_params = list(args[length:])
with torch.no_grad():
output_tensors = ctx.run_function(*ctx.input_tensors)
return output_tensors
@staticmethod
@torch.cuda.amp.custom_bwd # add this
def backward(ctx, *output_grads):
'''
for x in ctx.input_tensors:
if isinstance(x, int):
print('-----------------', ctx.run_function)
'''
ctx.input_tensors = [x.detach().requires_grad_(True) for x in ctx.input_tensors]
with torch.enable_grad():
# Fixes a bug where the first op in run_function modifies the
# Tensor storage in place, which is not allowed for detach()'d
# Tensors.
shallow_copies = [x.view_as(x) for x in ctx.input_tensors]
output_tensors = ctx.run_function(*shallow_copies)
input_grads = torch.autograd.grad(
output_tensors,
ctx.input_tensors + ctx.input_params,
output_grads,
allow_unused=True,
)
del ctx.input_tensors
del ctx.input_params
del output_tensors
return (None, None) + input_grads
+95
View File
@@ -0,0 +1,95 @@
import torch
import numpy as np
class AbstractDistribution:
def sample(self):
raise NotImplementedError()
def mode(self):
raise NotImplementedError()
class DiracDistribution(AbstractDistribution):
def __init__(self, value):
self.value = value
def sample(self):
return self.value
def mode(self):
return self.value
class DiagonalGaussianDistribution(object):
def __init__(self, parameters, deterministic=False):
self.parameters = parameters
self.mean, self.logvar = torch.chunk(parameters, 2, dim=1)
self.logvar = torch.clamp(self.logvar, -30.0, 20.0)
self.deterministic = deterministic
self.std = torch.exp(0.5 * self.logvar)
self.var = torch.exp(self.logvar)
if self.deterministic:
self.var = self.std = torch.zeros_like(self.mean).to(device=self.parameters.device)
def sample(self, noise=None):
if noise is None:
noise = torch.randn(self.mean.shape)
x = self.mean + self.std * noise.to(device=self.parameters.device)
return x
def kl(self, other=None):
if self.deterministic:
return torch.Tensor([0.])
else:
if other is None:
return 0.5 * torch.sum(torch.pow(self.mean, 2)
+ self.var - 1.0 - self.logvar,
dim=[1, 2, 3])
else:
return 0.5 * torch.sum(
torch.pow(self.mean - other.mean, 2) / other.var
+ self.var / other.var - 1.0 - self.logvar + other.logvar,
dim=[1, 2, 3])
def nll(self, sample, dims=[1,2,3]):
if self.deterministic:
return torch.Tensor([0.])
logtwopi = np.log(2.0 * np.pi)
return 0.5 * torch.sum(
logtwopi + self.logvar + torch.pow(sample - self.mean, 2) / self.var,
dim=dims)
def mode(self):
return self.mean
def normal_kl(mean1, logvar1, mean2, logvar2):
"""
source: https://github.com/openai/guided-diffusion/blob/27c20a8fab9cb472df5d6bdd6c8d11c8f430b924/guided_diffusion/losses.py#L12
Compute the KL divergence between two gaussians.
Shapes are automatically broadcasted, so batches can be compared to
scalars, among other use cases.
"""
tensor = None
for obj in (mean1, logvar1, mean2, logvar2):
if isinstance(obj, torch.Tensor):
tensor = obj
break
assert tensor is not None, "at least one argument must be a Tensor"
# Force variances to be Tensors. Broadcasting helps convert scalars to
# Tensors, but it does not work for torch.exp().
logvar1, logvar2 = [
x if isinstance(x, torch.Tensor) else torch.tensor(x).to(tensor)
for x in (logvar1, logvar2)
]
return 0.5 * (
-1.0
+ logvar2
- logvar1
+ torch.exp(logvar1 - logvar2)
+ ((mean1 - mean2) ** 2) * torch.exp(-logvar2)
)
+76
View File
@@ -0,0 +1,76 @@
import torch
from torch import nn
class LitEma(nn.Module):
def __init__(self, model, decay=0.9999, use_num_upates=True):
super().__init__()
if decay < 0.0 or decay > 1.0:
raise ValueError('Decay must be between 0 and 1')
self.m_name2s_name = {}
self.register_buffer('decay', torch.tensor(decay, dtype=torch.float32))
self.register_buffer('num_updates', torch.tensor(0,dtype=torch.int) if use_num_upates
else torch.tensor(-1,dtype=torch.int))
for name, p in model.named_parameters():
if p.requires_grad:
#remove as '.'-character is not allowed in buffers
s_name = name.replace('.','')
self.m_name2s_name.update({name:s_name})
self.register_buffer(s_name,p.clone().detach().data)
self.collected_params = []
def forward(self,model):
decay = self.decay
if self.num_updates >= 0:
self.num_updates += 1
decay = min(self.decay,(1 + self.num_updates) / (10 + self.num_updates))
one_minus_decay = 1.0 - decay
with torch.no_grad():
m_param = dict(model.named_parameters())
shadow_params = dict(self.named_buffers())
for key in m_param:
if m_param[key].requires_grad:
sname = self.m_name2s_name[key]
shadow_params[sname] = shadow_params[sname].type_as(m_param[key])
shadow_params[sname].sub_(one_minus_decay * (shadow_params[sname] - m_param[key]))
else:
assert not key in self.m_name2s_name
def copy_to(self, model):
m_param = dict(model.named_parameters())
shadow_params = dict(self.named_buffers())
for key in m_param:
if m_param[key].requires_grad:
m_param[key].data.copy_(shadow_params[self.m_name2s_name[key]].data)
else:
assert not key in self.m_name2s_name
def store(self, parameters):
"""
Save the current parameters for restoring later.
Args:
parameters: Iterable of `torch.nn.Parameter`; the parameters to be
temporarily stored.
"""
self.collected_params = [param.clone() for param in parameters]
def restore(self, parameters):
"""
Restore the parameters stored with the `store` method.
Useful to validate the model with EMA parameters without affecting the
original optimization process. Store the parameters before the
`copy_to` method. After validation (or model saving), use this to
restore the former parameters.
Args:
parameters: Iterable of `torch.nn.Parameter`; the parameters to be
updated with the stored parameters.
"""
for c_param, param in zip(self.collected_params, parameters):
param.data.copy_(c_param.data)
+221
View File
@@ -0,0 +1,221 @@
import os
from contextlib import contextmanager
import numpy as np
import pytorch_lightning as pl
import torch
import torch.nn.functional as F
from einops import rearrange
from lvdm.distributions import DiagonalGaussianDistribution
from lvdm.modules.networks.ae_modules import Decoder, Encoder
from utils.utils import instantiate_from_config
class AutoencoderKL(pl.LightningModule):
def __init__(self,
ddconfig,
lossconfig,
embed_dim,
ckpt_path=None,
ignore_keys=[],
image_key="image",
colorize_nlabels=None,
monitor=None,
test=False,
logdir=None,
input_dim=4,
test_args=None,
):
super().__init__()
self.image_key = image_key
self.encoder = Encoder(**ddconfig)
self.decoder = Decoder(**ddconfig)
self.loss = instantiate_from_config(lossconfig)
assert ddconfig["double_z"]
self.quant_conv = torch.nn.Conv2d(2*ddconfig["z_channels"], 2*embed_dim, 1)
self.post_quant_conv = torch.nn.Conv2d(embed_dim, ddconfig["z_channels"], 1)
self.embed_dim = embed_dim
self.input_dim = input_dim
self.test = test
self.test_args = test_args
self.logdir = logdir
if colorize_nlabels is not None:
assert type(colorize_nlabels)==int
self.register_buffer("colorize", torch.randn(3, colorize_nlabels, 1, 1))
if monitor is not None:
self.monitor = monitor
if ckpt_path is not None:
self.init_from_ckpt(ckpt_path, ignore_keys=ignore_keys)
if self.test:
self.init_test()
def init_test(self,):
self.test = True
save_dir = os.path.join(self.logdir, "test")
if 'ckpt' in self.test_args:
ckpt_name = os.path.basename(self.test_args.ckpt).split('.ckpt')[0] + f'_epoch{self._cur_epoch}'
self.root = os.path.join(save_dir, ckpt_name)
else:
self.root = save_dir
if 'test_subdir' in self.test_args:
self.root = os.path.join(save_dir, self.test_args.test_subdir)
self.root_zs = os.path.join(self.root, "zs")
self.root_dec = os.path.join(self.root, "reconstructions")
self.root_inputs = os.path.join(self.root, "inputs")
os.makedirs(self.root, exist_ok=True)
if self.test_args.save_z:
os.makedirs(self.root_zs, exist_ok=True)
if self.test_args.save_reconstruction:
os.makedirs(self.root_dec, exist_ok=True)
if self.test_args.save_input:
os.makedirs(self.root_inputs, exist_ok=True)
assert(self.test_args is not None)
self.test_maximum = getattr(self.test_args, 'test_maximum', None)
self.count = 0
self.eval_metrics = {}
self.decodes = []
self.save_decode_samples = 2048
def init_from_ckpt(self, path, ignore_keys=list()):
sd = torch.load(path, map_location="cpu")
try:
self._cur_epoch = sd['epoch']
sd = sd["state_dict"]
except:
self._cur_epoch = 'null'
keys = list(sd.keys())
for k in keys:
for ik in ignore_keys:
if k.startswith(ik):
print("Deleting key {} from state_dict.".format(k))
del sd[k]
self.load_state_dict(sd, strict=False)
# self.load_state_dict(sd, strict=True)
print(f"Restored from {path}")
def encode(self, x, **kwargs):
h = self.encoder(x)
moments = self.quant_conv(h)
posterior = DiagonalGaussianDistribution(moments)
return posterior
def decode(self, z, **kwargs):
z = self.post_quant_conv(z)
dec = self.decoder(z)
return dec
def forward(self, input, sample_posterior=True):
posterior = self.encode(input)
if sample_posterior:
z = posterior.sample()
else:
z = posterior.mode()
dec = self.decode(z)
return dec, posterior
def get_input(self, batch, k):
x = batch[k]
if x.dim() == 5 and self.input_dim == 4:
b,c,t,h,w = x.shape
self.b = b
self.t = t
x = rearrange(x, 'b c t h w -> (b t) c h w')
return x
def training_step(self, batch, batch_idx, optimizer_idx):
inputs = self.get_input(batch, self.image_key)
reconstructions, posterior = self(inputs)
if optimizer_idx == 0:
# train encoder+decoder+logvar
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("aeloss", aeloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_ae, prog_bar=False, logger=True, on_step=True, on_epoch=False)
return aeloss
if optimizer_idx == 1:
# train the discriminator
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("discloss", discloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_disc, prog_bar=False, logger=True, on_step=True, on_epoch=False)
return discloss
def validation_step(self, batch, batch_idx):
inputs = self.get_input(batch, self.image_key)
reconstructions, posterior = self(inputs)
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, 0, self.global_step,
last_layer=self.get_last_layer(), split="val")
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, 1, self.global_step,
last_layer=self.get_last_layer(), split="val")
self.log("val/rec_loss", log_dict_ae["val/rec_loss"])
self.log_dict(log_dict_ae)
self.log_dict(log_dict_disc)
return self.log_dict
def configure_optimizers(self):
lr = self.learning_rate
opt_ae = torch.optim.Adam(list(self.encoder.parameters())+
list(self.decoder.parameters())+
list(self.quant_conv.parameters())+
list(self.post_quant_conv.parameters()),
lr=lr, betas=(0.5, 0.9))
opt_disc = torch.optim.Adam(self.loss.discriminator.parameters(),
lr=lr, betas=(0.5, 0.9))
return [opt_ae, opt_disc], []
def get_last_layer(self):
return self.decoder.conv_out.weight
@torch.no_grad()
def log_images(self, batch, only_inputs=False, **kwargs):
log = dict()
x = self.get_input(batch, self.image_key)
x = x.to(self.device)
if not only_inputs:
xrec, posterior = self(x)
if x.shape[1] > 3:
# colorize with random projection
assert xrec.shape[1] > 3
x = self.to_rgb(x)
xrec = self.to_rgb(xrec)
log["samples"] = self.decode(torch.randn_like(posterior.sample()))
log["reconstructions"] = xrec
log["inputs"] = x
return log
def to_rgb(self, x):
assert self.image_key == "segmentation"
if not hasattr(self, "colorize"):
self.register_buffer("colorize", torch.randn(3, x.shape[1], 1, 1).to(x))
x = F.conv2d(x, weight=self.colorize)
x = 2.*(x-x.min())/(x.max()-x.min()) - 1.
return x
class IdentityFirstStage(torch.nn.Module):
def __init__(self, *args, vq_interface=False, **kwargs):
self.vq_interface = vq_interface # TODO: Should be true by default but check to not break older stuff
super().__init__()
def encode(self, x, *args, **kwargs):
return x
def decode(self, x, *args, **kwargs):
return x
def quantize(self, x, *args, **kwargs):
if self.vq_interface:
return x, None, [None, None, None]
return x
def forward(self, x, *args, **kwargs):
return x
File diff suppressed because it is too large Load Diff
View File
+283
View File
@@ -0,0 +1,283 @@
"""SAMPLING ONLY."""
import numpy as np
import torch
from tqdm import tqdm
from lvdm.common import noise_like
from lvdm.models.utils_diffusion import (make_ddim_sampling_parameters,
make_ddim_timesteps)
class DDIMSampler(object):
def __init__(self, model, schedule="linear", **kwargs):
super().__init__()
self.model = model
self.ddpm_num_timesteps = model.num_timesteps
self.schedule = schedule
self.counter = 0
def register_buffer(self, name, attr):
if type(attr) == torch.Tensor:
if attr.device != torch.device("cuda"):
attr = attr.to(torch.device("cuda"))
setattr(self, name, attr)
def make_schedule(self, ddim_num_steps, ddim_discretize="uniform", ddim_eta=0., verbose=True):
self.ddim_timesteps = make_ddim_timesteps(ddim_discr_method=ddim_discretize, num_ddim_timesteps=ddim_num_steps,
num_ddpm_timesteps=self.ddpm_num_timesteps,verbose=verbose)
alphas_cumprod = self.model.alphas_cumprod
assert alphas_cumprod.shape[0] == self.ddpm_num_timesteps, 'alphas have to be defined for each timestep'
to_torch = lambda x: x.clone().detach().to(torch.float32).to(self.model.device)
self.register_buffer('betas', to_torch(self.model.betas))
self.register_buffer('alphas_cumprod', to_torch(alphas_cumprod))
self.register_buffer('alphas_cumprod_prev', to_torch(self.model.alphas_cumprod_prev))
# calculations for diffusion q(x_t | x_{t-1}) and others
self.register_buffer('sqrt_alphas_cumprod', to_torch(np.sqrt(alphas_cumprod.cpu())))
self.register_buffer('sqrt_one_minus_alphas_cumprod', to_torch(np.sqrt(1. - alphas_cumprod.cpu())))
self.register_buffer('log_one_minus_alphas_cumprod', to_torch(np.log(1. - alphas_cumprod.cpu())))
self.register_buffer('sqrt_recip_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod.cpu())))
self.register_buffer('sqrt_recipm1_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod.cpu() - 1)))
# ddim sampling parameters
ddim_sigmas, ddim_alphas, ddim_alphas_prev = make_ddim_sampling_parameters(alphacums=alphas_cumprod.cpu(),
ddim_timesteps=self.ddim_timesteps,
eta=ddim_eta,verbose=verbose)
self.register_buffer('ddim_sigmas', ddim_sigmas)
self.register_buffer('ddim_alphas', ddim_alphas)
self.register_buffer('ddim_alphas_prev', ddim_alphas_prev)
self.register_buffer('ddim_sqrt_one_minus_alphas', np.sqrt(1. - ddim_alphas))
sigmas_for_original_sampling_steps = ddim_eta * torch.sqrt(
(1 - self.alphas_cumprod_prev) / (1 - self.alphas_cumprod) * (
1 - self.alphas_cumprod / self.alphas_cumprod_prev))
self.register_buffer('ddim_sigmas_for_original_num_steps', sigmas_for_original_sampling_steps)
@torch.no_grad()
def sample(self,
S,
batch_size,
shape,
conditioning=None,
callback=None,
normals_sequence=None,
img_callback=None,
quantize_x0=False,
eta=0.,
mask=None,
x0=None,
temperature=1.,
noise_dropout=0.,
score_corrector=None,
corrector_kwargs=None,
verbose=True,
schedule_verbose=False,
x_T=None,
log_every_t=100,
unconditional_guidance_scale=1.,
unconditional_conditioning=None,
# this has to come in the same format as the conditioning, # e.g. as encoded tokens, ...
**kwargs
):
# check condition bs
if conditioning is not None:
if isinstance(conditioning, dict):
try:
cbs = conditioning[list(conditioning.keys())[0]].shape[0]
except:
cbs = conditioning[list(conditioning.keys())[0]][0].shape[0]
if cbs != batch_size:
print(f"Warning: Got {cbs} conditionings but batch-size is {batch_size}")
else:
if conditioning.shape[0] != batch_size:
print(f"Warning: Got {conditioning.shape[0]} conditionings but batch-size is {batch_size}")
self.make_schedule(ddim_num_steps=S, ddim_eta=eta, verbose=schedule_verbose)
# make shape
if len(shape) == 3:
C, H, W = shape
size = (batch_size, C, H, W)
elif len(shape) == 4:
C, T, H, W = shape
size = (batch_size, C, T, H, W)
# print(f'Data shape for DDIM sampling is {size}, eta {eta}')
samples, intermediates = self.ddim_sampling(conditioning, size,
callback=callback,
img_callback=img_callback,
quantize_denoised=quantize_x0,
mask=mask, x0=x0,
ddim_use_original_steps=False,
noise_dropout=noise_dropout,
temperature=temperature,
score_corrector=score_corrector,
corrector_kwargs=corrector_kwargs,
x_T=x_T,
log_every_t=log_every_t,
unconditional_guidance_scale=unconditional_guidance_scale,
unconditional_conditioning=unconditional_conditioning,
verbose=verbose,
**kwargs)
return samples, intermediates
@torch.no_grad()
def ddim_sampling(self, cond, shape,
x_T=None, ddim_use_original_steps=False,
callback=None, timesteps=None, quantize_denoised=False,
mask=None, x0=None, img_callback=None, log_every_t=100,
temperature=1., noise_dropout=0., score_corrector=None, corrector_kwargs=None,
unconditional_guidance_scale=1., unconditional_conditioning=None, verbose=True,
**kwargs):
device = self.model.betas.device
b = shape[0]
if x_T is None:
img = torch.randn(shape, device=device)
else:
img = x_T
if timesteps is None:
timesteps = self.ddpm_num_timesteps if ddim_use_original_steps else self.ddim_timesteps
elif timesteps is not None and not ddim_use_original_steps:
subset_end = int(min(timesteps / self.ddim_timesteps.shape[0], 1) * self.ddim_timesteps.shape[0]) - 1
timesteps = self.ddim_timesteps[:subset_end]
intermediates = {'x_inter': [img], 'pred_x0': [img]}
time_range = reversed(range(0,timesteps)) if ddim_use_original_steps else np.flip(timesteps)
total_steps = timesteps if ddim_use_original_steps else timesteps.shape[0]
if verbose:
iterator = tqdm(time_range, desc='DDIM Sampler', total=total_steps)
else:
iterator = time_range
clean_cond = kwargs.pop("clean_cond", False)
for i, step in enumerate(iterator):
index = total_steps - i - 1
ts = torch.full((b,), step, device=device, dtype=torch.long)
# use mask to blend noised original latent (img_orig) & new sampled latent (img)
if mask is not None:
assert x0 is not None
if clean_cond:
img_orig = x0
else:
img_orig = self.model.q_sample(x0, ts) # TODO: deterministic forward pass? <ddim inversion>
img = img_orig * mask + (1. - mask) * img # keep original & modify use img
outs = self.p_sample_ddim(img, cond, ts, index=index, use_original_steps=ddim_use_original_steps,
quantize_denoised=quantize_denoised, temperature=temperature,
noise_dropout=noise_dropout, score_corrector=score_corrector,
corrector_kwargs=corrector_kwargs,
unconditional_guidance_scale=unconditional_guidance_scale,
unconditional_conditioning=unconditional_conditioning,
**kwargs)
img, pred_x0 = outs
if callback: callback(i)
if img_callback: img_callback(pred_x0, i)
if index % log_every_t == 0 or index == total_steps - 1:
intermediates['x_inter'].append(img)
intermediates['pred_x0'].append(pred_x0)
return img, intermediates
@torch.no_grad()
def p_sample_ddim(self, x, c, t, index, repeat_noise=False, use_original_steps=False, quantize_denoised=False,
temperature=1., noise_dropout=0., score_corrector=None, corrector_kwargs=None,
unconditional_guidance_scale=1., unconditional_conditioning=None,
uc_type=None, conditional_guidance_scale_temporal=None, **kwargs):
b, *_, device = *x.shape, x.device
if x.dim() == 5:
is_video = True
else:
is_video = False
# f=open('/apdcephfs_cq2/share_1290939/yingqinghe/code/LVDM-private/cfg_range_s5noclamp.txt','a')
# print(f't={t}, model input, min={torch.min(x)}, max={torch.max(x)}',file=f)
if unconditional_conditioning is None or unconditional_guidance_scale == 1.:
e_t = self.model.apply_model(x, t, c, **kwargs) # unet denoiser
else:
# with unconditional condition
if isinstance(c, torch.Tensor):
un_kwargs = kwargs.copy()
if isinstance(unconditional_conditioning, dict):
for uk, uv in unconditional_conditioning.items():
if uk in un_kwargs:
un_kwargs[uk] = uv
unconditional_conditioning = unconditional_conditioning['uc']
if 'cond_T' in kwargs and t < kwargs['cond_T']:
if 'features_adapter' in kwargs:
kwargs.pop('features_adapter')
un_kwargs.pop('features_adapter')
# kwargs['features_adapter'] = None
# un_kwargs['features_adapter'] = None
# if 'pose_emb' in kwargs:
# kwargs.pop('pose_emb')
# un_kwargs.pop('pose_emb')
# kwargs['pose_emb'] = None
# un_kwargs['pose_emb'] = None
e_t = self.model.apply_model(x, t, c, **kwargs)
# e_t_uncond = self.model.apply_model(x, t, unconditional_conditioning, **kwargs)
e_t_uncond = self.model.apply_model(x, t, unconditional_conditioning, **un_kwargs)
elif isinstance(c, dict):
e_t = self.model.apply_model(x, t, c, **kwargs)
e_t_uncond = self.model.apply_model(x, t, unconditional_conditioning, **kwargs)
else:
raise NotImplementedError
# text cfg
if uc_type is None:
e_t = e_t_uncond + unconditional_guidance_scale * (e_t - e_t_uncond)
else:
if uc_type == 'cfg_original':
e_t = e_t + unconditional_guidance_scale * (e_t - e_t_uncond)
elif uc_type == 'cfg_ours':
e_t = e_t + unconditional_guidance_scale * (e_t_uncond - e_t)
else:
raise NotImplementedError
# temporal guidance
if conditional_guidance_scale_temporal is not None:
e_t_temporal = self.model.apply_model(x, t, c, **kwargs)
e_t_image = self.model.apply_model(x, t, c, no_temporal_attn=True, **kwargs)
e_t = e_t + conditional_guidance_scale_temporal * (e_t_temporal - e_t_image)
if score_corrector is not None:
assert self.model.parameterization == "eps"
e_t = score_corrector.modify_score(self.model, e_t, x, t, c, **corrector_kwargs)
alphas = self.model.alphas_cumprod if use_original_steps else self.ddim_alphas
alphas_prev = self.model.alphas_cumprod_prev if use_original_steps else self.ddim_alphas_prev
sqrt_one_minus_alphas = self.model.sqrt_one_minus_alphas_cumprod if use_original_steps else self.ddim_sqrt_one_minus_alphas
sigmas = self.model.ddim_sigmas_for_original_num_steps if use_original_steps else self.ddim_sigmas
# select parameters corresponding to the currently considered timestep
if is_video:
size = (b, 1, 1, 1, 1)
else:
size = (b, 1, 1, 1)
a_t = torch.full(size, alphas[index], device=device)
a_prev = torch.full(size, alphas_prev[index], device=device)
sigma_t = torch.full(size, sigmas[index], device=device)
sqrt_one_minus_at = torch.full(size, sqrt_one_minus_alphas[index],device=device)
# current prediction for x_0
pred_x0 = (x - sqrt_one_minus_at * e_t) / a_t.sqrt()
# print(f't={t}, pred_x0, min={torch.min(pred_x0)}, max={torch.max(pred_x0)}',file=f)
if quantize_denoised:
pred_x0, _, *_ = self.model.first_stage_model.quantize(pred_x0)
# direction pointing to x_t
dir_xt = (1. - a_prev - sigma_t**2).sqrt() * e_t
# # norm pred_x0
# p=2
# s=()
# pred_x0 = pred_x0 - torch.max(torch.abs(pred_x0))
noise = sigma_t * noise_like(x.shape, device, repeat_noise) * temperature
if noise_dropout > 0.:
noise = torch.nn.functional.dropout(noise, p=noise_dropout)
x_prev = a_prev.sqrt() * pred_x0 + dir_xt + noise
return x_prev, pred_x0
+105
View File
@@ -0,0 +1,105 @@
import math
import numpy as np
from einops import repeat
import torch
import torch.nn.functional as F
def timestep_embedding(timesteps, dim, max_period=10000, repeat_only=False):
"""
Create sinusoidal timestep embeddings.
:param timesteps: a 1-D Tensor of N indices, one per batch element.
These may be fractional.
:param dim: the dimension of the output.
:param max_period: controls the minimum frequency of the embeddings.
:return: an [N x dim] Tensor of positional embeddings.
"""
if not repeat_only:
half = dim // 2
freqs = torch.exp(
-math.log(max_period) * torch.arange(start=0, end=half, dtype=torch.float32) / half
).to(device=timesteps.device)
args = timesteps[:, None].float() * freqs[None]
embedding = torch.cat([torch.cos(args), torch.sin(args)], dim=-1)
if dim % 2:
embedding = torch.cat([embedding, torch.zeros_like(embedding[:, :1])], dim=-1)
else:
embedding = repeat(timesteps, 'b -> b d', d=dim)
return embedding
def make_beta_schedule(schedule, n_timestep, linear_start=1e-4, linear_end=2e-2, cosine_s=8e-3):
if schedule == "linear":
betas = (
torch.linspace(linear_start ** 0.5, linear_end ** 0.5, n_timestep, dtype=torch.float64) ** 2
)
elif schedule == "cosine":
timesteps = (
torch.arange(n_timestep + 1, dtype=torch.float64) / n_timestep + cosine_s
)
alphas = timesteps / (1 + cosine_s) * np.pi / 2
alphas = torch.cos(alphas).pow(2)
alphas = alphas / alphas[0]
betas = 1 - alphas[1:] / alphas[:-1]
betas = np.clip(betas, a_min=0, a_max=0.999)
elif schedule == "sqrt_linear":
betas = torch.linspace(linear_start, linear_end, n_timestep, dtype=torch.float64)
elif schedule == "sqrt":
betas = torch.linspace(linear_start, linear_end, n_timestep, dtype=torch.float64) ** 0.5
else:
raise ValueError(f"schedule '{schedule}' unknown.")
return betas.numpy()
def make_ddim_timesteps(ddim_discr_method, num_ddim_timesteps, num_ddpm_timesteps, verbose=True):
if ddim_discr_method == 'uniform':
c = num_ddpm_timesteps // num_ddim_timesteps
ddim_timesteps = np.asarray(list(range(0, num_ddpm_timesteps, c)))
elif ddim_discr_method == 'quad':
ddim_timesteps = ((np.linspace(0, np.sqrt(num_ddpm_timesteps * .8), num_ddim_timesteps)) ** 2).astype(int)
else:
raise NotImplementedError(f'There is no ddim discretization method called "{ddim_discr_method}"')
# assert ddim_timesteps.shape[0] == num_ddim_timesteps
# add one to get the final alpha values right (the ones from first scale to data during sampling)
steps_out = ddim_timesteps + 1
if verbose:
print(f'Selected timesteps for ddim sampler: {steps_out}')
return steps_out
def make_ddim_sampling_parameters(alphacums, ddim_timesteps, eta, verbose=True):
# select alphas for computing the variance schedule
# print(f'ddim_timesteps={ddim_timesteps}, len_alphacums={len(alphacums)}')
alphas = alphacums[ddim_timesteps]
alphas_prev = np.asarray([alphacums[0]] + alphacums[ddim_timesteps[:-1]].tolist())
# according the the formula provided in https://arxiv.org/abs/2010.02502
sigmas = eta * np.sqrt((1 - alphas_prev) / (1 - alphas) * (1 - alphas / alphas_prev))
if verbose:
print(f'Selected alphas for ddim sampler: a_t: {alphas}; a_(t-1): {alphas_prev}')
print(f'For the chosen value of eta, which is {eta}, '
f'this results in the following sigma_t schedule for ddim sampler {sigmas}')
return sigmas, alphas, alphas_prev
def betas_for_alpha_bar(num_diffusion_timesteps, alpha_bar, max_beta=0.999):
"""
Create a beta schedule that discretizes the given alpha_t_bar function,
which defines the cumulative product of (1-beta) over time from t = [0,1].
:param num_diffusion_timesteps: the number of betas to produce.
:param alpha_bar: a lambda that takes an argument t from 0 to 1 and
produces the cumulative product of (1-beta) up to that
part of the diffusion process.
:param max_beta: the maximum beta to use; use values lower than 1 to
prevent singularities.
"""
betas = []
for i in range(num_diffusion_timesteps):
t1 = i / num_diffusion_timesteps
t2 = (i + 1) / num_diffusion_timesteps
betas.append(min(1 - alpha_bar(t2) / alpha_bar(t1), max_beta))
return np.array(betas)
+428
View File
@@ -0,0 +1,428 @@
import math
from functools import partial
from inspect import isfunction
import torch
import torch.nn.functional as F
from einops import rearrange, repeat
from torch import einsum, nn
try:
import xformers
import xformers.ops
XFORMERS_IS_AVAILBLE = True
except:
XFORMERS_IS_AVAILBLE = False
from lvdm.basics import conv_nd, normalization, zero_module
from lvdm.common import checkpoint, default, exists, init_, max_neg_value, uniq
class RelativePosition(nn.Module):
""" https://github.com/evelinehong/Transformer_Relative_Position_PyTorch/blob/master/relative_position.py """
def __init__(self, num_units, max_relative_position):
super().__init__()
self.num_units = num_units
self.max_relative_position = max_relative_position
self.embeddings_table = nn.Parameter(torch.Tensor(max_relative_position * 2 + 1, num_units))
nn.init.xavier_uniform_(self.embeddings_table)
def forward(self, length_q, length_k):
device = self.embeddings_table.device
range_vec_q = torch.arange(length_q, device=device)
range_vec_k = torch.arange(length_k, device=device)
distance_mat = range_vec_k[None, :] - range_vec_q[:, None]
distance_mat_clipped = torch.clamp(distance_mat, -self.max_relative_position, self.max_relative_position)
final_mat = distance_mat_clipped + self.max_relative_position
# final_mat = th.LongTensor(final_mat).to(self.embeddings_table.device)
# final_mat = th.tensor(final_mat, device=self.embeddings_table.device, dtype=torch.long)
final_mat = final_mat.long()
embeddings = self.embeddings_table[final_mat]
return embeddings
class CrossAttention(nn.Module):
def __init__(self, query_dim, context_dim=None, heads=8, dim_head=64, dropout=0.,
relative_position=False, temporal_length=None):
super().__init__()
inner_dim = dim_head * heads
context_dim = default(context_dim, query_dim)
self.scale = dim_head**-0.5
self.heads = heads
self.dim_head = dim_head
self.to_q = nn.Linear(query_dim, inner_dim, bias=False)
self.to_k = nn.Linear(context_dim, inner_dim, bias=False)
self.to_v = nn.Linear(context_dim, inner_dim, bias=False)
self.to_out = nn.Sequential(nn.Linear(inner_dim, query_dim), nn.Dropout(dropout))
self.relative_position = relative_position
if self.relative_position:
assert(temporal_length is not None)
self.relative_position_k = RelativePosition(num_units=dim_head, max_relative_position=temporal_length)
self.relative_position_v = RelativePosition(num_units=dim_head, max_relative_position=temporal_length)
else:
## only used for spatial attention, while NOT for temporal attention
if XFORMERS_IS_AVAILBLE and temporal_length is None:
self.forward = self.efficient_forward
def forward(self, x, context=None, mask=None):
h = self.heads
q = self.to_q(x)
context = default(context, x)
k = self.to_k(context)
v = self.to_v(context)
q, k, v = map(lambda t: rearrange(t, 'b n (h d) -> (b h) n d', h=h), (q, k, v))
sim = torch.einsum('b i d, b j d -> b i j', q, k) * self.scale
if self.relative_position:
len_q, len_k, len_v = q.shape[1], k.shape[1], v.shape[1]
k2 = self.relative_position_k(len_q, len_k)
sim2 = einsum('b t d, t s d -> b t s', q, k2) * self.scale # TODO check
sim += sim2
del q, k
if exists(mask):
## feasible for causal attention mask only
max_neg_value = -torch.finfo(sim.dtype).max
mask = repeat(mask, 'b i j -> (b h) i j', h=h)
sim.masked_fill_(~(mask>0.5), max_neg_value)
# attention, what we cannot get enough of
sim = sim.softmax(dim=-1)
out = torch.einsum('b i j, b j d -> b i d', sim, v)
if self.relative_position:
v2 = self.relative_position_v(len_q, len_v)
out2 = einsum('b t s, t s d -> b t d', sim, v2) # TODO check
out += out2
out = rearrange(out, '(b h) n d -> b n (h d)', h=h)
return self.to_out(out)
def efficient_forward(self, x, context=None, mask=None):
q = self.to_q(x)
context = default(context, x)
k = self.to_k(context)
v = self.to_v(context)
b, _, _ = q.shape
q, k, v = map(
lambda t: t.unsqueeze(3)
.reshape(b, t.shape[1], self.heads, self.dim_head)
.permute(0, 2, 1, 3)
.reshape(b * self.heads, t.shape[1], self.dim_head)
.contiguous(),
(q, k, v),
)
# actually compute the attention, what we cannot get enough of
out = xformers.ops.memory_efficient_attention(q, k, v, attn_bias=None, op=None)
if exists(mask):
raise NotImplementedError
out = (
out.unsqueeze(0)
.reshape(b, self.heads, out.shape[1], self.dim_head)
.permute(0, 2, 1, 3)
.reshape(b, out.shape[1], self.heads * self.dim_head)
)
return self.to_out(out)
class BasicTransformerBlock(nn.Module):
def __init__(self, dim, n_heads, d_head, dropout=0., context_dim=None, gated_ff=True, checkpoint=True,
disable_self_attn=False, attention_cls=None):
super().__init__()
attn_cls = CrossAttention if attention_cls is None else attention_cls
self.disable_self_attn = disable_self_attn
self.attn1 = attn_cls(query_dim=dim, heads=n_heads, dim_head=d_head, dropout=dropout,
context_dim=context_dim if self.disable_self_attn else None)
self.ff = FeedForward(dim, dropout=dropout, glu=gated_ff)
self.attn2 = attn_cls(query_dim=dim, context_dim=context_dim, heads=n_heads, dim_head=d_head, dropout=dropout)
self.norm1 = nn.LayerNorm(dim)
self.norm2 = nn.LayerNorm(dim)
self.norm3 = nn.LayerNorm(dim)
self.checkpoint = checkpoint
def forward(self, x, context=None, mask=None):
## implementation tricks: because checkpointing doesn't support non-tensor (e.g. None or scalar) arguments
input_tuple = (x,) ## should not be (x), otherwise *input_tuple will decouple x into multiple arguments
if context is not None:
input_tuple = (x, context)
if mask is not None:
forward_mask = partial(self._forward, mask=mask)
return checkpoint(forward_mask, (x,), self.parameters(), self.checkpoint)
# It seems that it will not be executed forever
if context is not None and mask is not None:
input_tuple = (x, context, mask)
return checkpoint(self._forward, input_tuple, self.parameters(), self.checkpoint)
def _forward(self, x, context=None, mask=None):
x = self.attn1(self.norm1(x), context=context if self.disable_self_attn else None, mask=mask) + x
x = self.attn2(self.norm2(x), context=context, mask=mask) + x
x = self.ff(self.norm3(x)) + x
return x
class SpatialTransformer(nn.Module):
"""
Transformer block for image-like data in spatial axis.
First, project the input (aka embedding)
and reshape to b, t, d.
Then apply standard transformer action.
Finally, reshape to image
NEW: use_linear for more efficiency instead of the 1x1 convs
"""
def __init__(self, in_channels, n_heads, d_head, depth=1, dropout=0., context_dim=None,
use_checkpoint=True, disable_self_attn=False, use_linear=False):
super().__init__()
self.in_channels = in_channels
inner_dim = n_heads * d_head
self.norm = torch.nn.GroupNorm(num_groups=32, num_channels=in_channels, eps=1e-6, affine=True)
if not use_linear:
self.proj_in = nn.Conv2d(in_channels, inner_dim, kernel_size=1, stride=1, padding=0)
else:
self.proj_in = nn.Linear(in_channels, inner_dim)
self.transformer_blocks = nn.ModuleList([
BasicTransformerBlock(
inner_dim,
n_heads,
d_head,
dropout=dropout,
context_dim=context_dim,
disable_self_attn=disable_self_attn,
checkpoint=use_checkpoint) for d in range(depth)
])
if not use_linear:
self.proj_out = zero_module(nn.Conv2d(inner_dim, in_channels, kernel_size=1, stride=1, padding=0))
else:
self.proj_out = zero_module(nn.Linear(inner_dim, in_channels))
self.use_linear = use_linear
def forward(self, x, context=None):
b, c, h, w = x.shape
x_in = x
x = self.norm(x)
if not self.use_linear:
x = self.proj_in(x)
x = rearrange(x, 'b c h w -> b (h w) c').contiguous()
if self.use_linear:
x = self.proj_in(x)
for i, block in enumerate(self.transformer_blocks):
x = block(x, context=context)
if self.use_linear:
x = self.proj_out(x)
x = rearrange(x, 'b (h w) c -> b c h w', h=h, w=w).contiguous()
if not self.use_linear:
x = self.proj_out(x)
return x + x_in
class TemporalTransformer(nn.Module):
"""
Transformer block for image-like data in temporal axis.
First, reshape to b, t, d.
Then apply standard transformer action.
Finally, reshape to image
"""
def __init__(self, in_channels, n_heads, d_head, depth=1, dropout=0., context_dim=None,
use_checkpoint=True, use_linear=False, only_self_att=True, causal_attention=False,
relative_position=False, temporal_length=None, use_image_dataset=False):
super().__init__()
self.only_self_att = only_self_att
self.relative_position = relative_position
self.causal_attention = causal_attention
self.use_image_dataset = use_image_dataset
self.in_channels = in_channels
inner_dim = n_heads * d_head
self.norm = torch.nn.GroupNorm(num_groups=32, num_channels=in_channels, eps=1e-6, affine=True)
self.proj_in = nn.Conv1d(in_channels, inner_dim, kernel_size=1, stride=1, padding=0)
if not use_linear:
self.proj_in = nn.Conv1d(in_channels, inner_dim, kernel_size=1, stride=1, padding=0)
else:
self.proj_in = nn.Linear(in_channels, inner_dim)
if relative_position:
assert(temporal_length is not None)
attention_cls = partial(CrossAttention, relative_position=True, temporal_length=temporal_length)
else:
attention_cls = None
if self.only_self_att:
context_dim = None
self.transformer_blocks = nn.ModuleList([
BasicTransformerBlock(
inner_dim,
n_heads,
d_head,
dropout=dropout,
context_dim=context_dim,
attention_cls=attention_cls,
checkpoint=use_checkpoint) for d in range(depth)
])
if not use_linear:
self.proj_out = zero_module(nn.Conv1d(inner_dim, in_channels, kernel_size=1, stride=1, padding=0))
else:
self.proj_out = zero_module(nn.Linear(inner_dim, in_channels))
self.use_linear = use_linear
def forward(self, x, context=None, is_imgbatch=False):
b, c, t, h, w = x.shape
x_in = x
x = self.norm(x)
x = rearrange(x, 'b c t h w -> (b h w) c t').contiguous()
if not self.use_linear:
x = self.proj_in(x)
x = rearrange(x, 'bhw c t -> bhw t c').contiguous()
if self.use_linear:
x = self.proj_in(x)
temp_mask = None
if self.causal_attention:
temp_mask = torch.tril(torch.ones([1, t, t]))
if is_imgbatch:
temp_mask = torch.eye(t).unsqueeze(0)
if temp_mask is not None:
mask = temp_mask.to(x.device)
mask = repeat(mask, 'l i j -> (l bhw) i j', bhw=b*h*w)
else:
mask = None
if self.only_self_att:
## note: if no context is given, cross-attention defaults to self-attention
for i, block in enumerate(self.transformer_blocks):
x = block(x, mask=mask)
x = rearrange(x, '(b hw) t c -> b hw t c', b=b).contiguous()
else:
x = rearrange(x, '(b hw) t c -> b hw t c', b=b).contiguous()
context = rearrange(context, '(b t) l con -> b t l con', t=t).contiguous()
for i, block in enumerate(self.transformer_blocks):
# calculate each batch one by one (since number in shape could not greater then 65,535 for some package)
for j in range(b):
unit_context = context[j][0:1]
context_j = repeat(unit_context, 't l con -> (t r) l con', r=(h * w)).contiguous()
## note: causal mask will not applied in cross-attention case
x[j] = block(x[j], context=context_j)
if self.use_linear:
x = self.proj_out(x)
x = rearrange(x, 'b (h w) t c -> b c t h w', h=h, w=w).contiguous()
if not self.use_linear:
x = rearrange(x, 'b hw t c -> (b hw) c t').contiguous()
x = self.proj_out(x)
x = rearrange(x, '(b h w) c t -> b c t h w', b=b, h=h, w=w).contiguous()
if self.use_image_dataset:
x = 0.0 * x + x_in
else:
x = x + x_in
return x
class GEGLU(nn.Module):
def __init__(self, dim_in, dim_out):
super().__init__()
self.proj = nn.Linear(dim_in, dim_out * 2)
def forward(self, x):
x, gate = self.proj(x).chunk(2, dim=-1)
return x * F.gelu(gate)
class FeedForward(nn.Module):
def __init__(self, dim, dim_out=None, mult=4, glu=False, dropout=0.):
super().__init__()
inner_dim = int(dim * mult)
dim_out = default(dim_out, dim)
project_in = nn.Sequential(
nn.Linear(dim, inner_dim),
nn.GELU()
) if not glu else GEGLU(dim, inner_dim)
self.net = nn.Sequential(
project_in,
nn.Dropout(dropout),
nn.Linear(inner_dim, dim_out)
)
def forward(self, x):
return self.net(x)
class LinearAttention(nn.Module):
def __init__(self, dim, heads=4, dim_head=32):
super().__init__()
self.heads = heads
hidden_dim = dim_head * heads
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
self.to_out = nn.Conv2d(hidden_dim, dim, 1)
def forward(self, x):
b, c, h, w = x.shape
qkv = self.to_qkv(x)
q, k, v = rearrange(qkv, 'b (qkv heads c) h w -> qkv b heads c (h w)', heads = self.heads, qkv=3)
k = k.softmax(dim=-1)
context = torch.einsum('bhdn,bhen->bhde', k, v)
out = torch.einsum('bhde,bhdn->bhen', context, q)
out = rearrange(out, 'b heads c (h w) -> b (heads c) h w', heads=self.heads, h=h, w=w)
return self.to_out(out)
class SpatialSelfAttention(nn.Module):
def __init__(self, in_channels):
super().__init__()
self.in_channels = in_channels
self.norm = torch.nn.GroupNorm(num_groups=32, num_channels=in_channels, eps=1e-6, affine=True)
self.q = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
self.k = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
self.v = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
self.proj_out = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
def forward(self, x):
h_ = x
h_ = self.norm(h_)
q = self.q(h_)
k = self.k(h_)
v = self.v(h_)
# compute attention
b,c,h,w = q.shape
q = rearrange(q, 'b c h w -> b (h w) c')
k = rearrange(k, 'b c h w -> b c (h w)')
w_ = torch.einsum('bij,bjk->bik', q, k)
w_ = w_ * (int(c)**(-0.5))
w_ = torch.nn.functional.softmax(w_, dim=2)
# attend to values
v = rearrange(v, 'b c h w -> b c (h w)')
w_ = rearrange(w_, 'b i j -> b j i')
h_ = torch.einsum('bij,bjk->bik', v, w_)
h_ = rearrange(h_, 'b c (h w) -> b c h w', h=h)
h_ = self.proj_out(h_)
return x+h_
+995
View File
@@ -0,0 +1,995 @@
import math
from inspect import isfunction
import torch
import torch as th
from torch import nn, einsum
import torch.nn.functional as F
from einops import rearrange, repeat
try:
import xformers
import xformers.ops
XFORMERS_IS_AVAILBLE = True
except:
XFORMERS_IS_AVAILBLE = False
from lvdm.common import (
checkpoint,
exists,
uniq,
default,
max_neg_value,
init_
)
from lvdm.basics import (
conv_nd,
zero_module,
normalization
)
class GEGLU(nn.Module):
def __init__(self, dim_in, dim_out):
super().__init__()
self.proj = nn.Linear(dim_in, dim_out * 2)
def forward(self, x):
x, gate = self.proj(x).chunk(2, dim=-1)
return x * F.gelu(gate)
class FeedForward(nn.Module):
def __init__(self, dim, dim_out=None, mult=4, glu=False, dropout=0.):
super().__init__()
inner_dim = int(dim * mult)
dim_out = default(dim_out, dim)
project_in = nn.Sequential(
nn.Linear(dim, inner_dim),
nn.GELU()
) if not glu else GEGLU(dim, inner_dim)
self.net = nn.Sequential(
project_in,
nn.Dropout(dropout),
nn.Linear(inner_dim, dim_out)
)
def forward(self, x):
return self.net(x)
def Normalize(in_channels):
return torch.nn.GroupNorm(num_groups=32, num_channels=in_channels, eps=1e-6, affine=True)
# ---------------------------------------------------------------------------------------------------
class RelativePosition(nn.Module):
""" https://github.com/evelinehong/Transformer_Relative_Position_PyTorch/blob/master/relative_position.py """
def __init__(self, num_units, max_relative_position):
super().__init__()
self.num_units = num_units
self.max_relative_position = max_relative_position
self.embeddings_table = nn.Parameter(th.Tensor(max_relative_position * 2 + 1, num_units))
nn.init.xavier_uniform_(self.embeddings_table)
def forward(self, length_q, length_k):
device = self.embeddings_table.device
range_vec_q = th.arange(length_q, device=device)
range_vec_k = th.arange(length_k, device=device)
distance_mat = range_vec_k[None, :] - range_vec_q[:, None]
distance_mat_clipped = th.clamp(distance_mat, -self.max_relative_position, self.max_relative_position)
final_mat = distance_mat_clipped + self.max_relative_position
# final_mat = th.LongTensor(final_mat).to(self.embeddings_table.device)
# final_mat = th.tensor(final_mat, device=self.embeddings_table.device, dtype=torch.long)
final_mat = final_mat.long()
embeddings = self.embeddings_table[final_mat]
return embeddings
class TemporalCrossAttention(nn.Module):
def __init__(self,
query_dim,
context_dim=None,
heads=8,
dim_head=64,
dropout=0.,
temporal_length=None, # For relative positional representation and image-video joint training.
image_length=None, # For image-video joint training.
use_relative_position=False, # whether use relative positional representation in temporal attention.
img_video_joint_train=False, # For image-video joint training.
use_tempoal_causal_attn=False,
bidirectional_causal_attn=False,
tempoal_attn_type=None,
joint_train_mode="same_batch",
**kwargs,
):
super().__init__()
inner_dim = dim_head * heads
context_dim = default(context_dim, query_dim)
self.context_dim = context_dim
self.scale = dim_head ** -0.5
self.heads = heads
self.temporal_length = temporal_length
self.use_relative_position = use_relative_position
self.img_video_joint_train = img_video_joint_train
self.bidirectional_causal_attn = bidirectional_causal_attn
self.joint_train_mode = joint_train_mode
assert(joint_train_mode in ["same_batch", "diff_batch"])
self.tempoal_attn_type = tempoal_attn_type
if bidirectional_causal_attn:
assert use_tempoal_causal_attn
if tempoal_attn_type:
assert(tempoal_attn_type in ['sparse_causal', 'sparse_causal_first'])
assert(not use_tempoal_causal_attn)
assert(not (img_video_joint_train and (self.joint_train_mode == "same_batch")))
self.to_q = nn.Linear(query_dim, inner_dim, bias=False)
self.to_k = nn.Linear(context_dim, inner_dim, bias=False)
self.to_v = nn.Linear(context_dim, inner_dim, bias=False)
assert(not (img_video_joint_train and (self.joint_train_mode == "same_batch") and use_tempoal_causal_attn))
if img_video_joint_train:
if self.joint_train_mode == "same_batch":
mask = torch.ones([1, temporal_length+image_length, temporal_length+image_length])
# mask[:, image_length:, :] = 0
# mask[:, :, image_length:] = 0
mask[:, temporal_length:, :] = 0
mask[:, :, temporal_length:] = 0
self.mask = mask
else:
self.mask = None
elif use_tempoal_causal_attn:
# normal causal attn
self.mask = torch.tril(torch.ones([1, temporal_length, temporal_length]))
elif tempoal_attn_type == 'sparse_causal':
# all frames interact with only the `prev` & self frame
mask1 = torch.tril(torch.ones([1, temporal_length, temporal_length])).bool() # true indicates keeping
mask2 = torch.zeros([1, temporal_length, temporal_length]) # initialize to same shape with mask1
mask2[:,2:temporal_length, :temporal_length-2] = torch.tril(torch.ones([1,temporal_length-2, temporal_length-2]))
mask2=(1-mask2).bool() # false indicates masking
self.mask = mask1 & mask2
elif tempoal_attn_type == 'sparse_causal_first':
# all frames interact with only the `first` & self frame
mask1 = torch.tril(torch.ones([1, temporal_length, temporal_length])).bool() # true indicates keeping
mask2 = torch.zeros([1, temporal_length, temporal_length])
mask2[:,2:temporal_length, 1:temporal_length-1] = torch.tril(torch.ones([1,temporal_length-2, temporal_length-2]))
mask2=(1-mask2).bool() # false indicates masking
self.mask = mask1 & mask2
else:
self.mask = None
if use_relative_position:
assert(temporal_length is not None)
self.relative_position_k = RelativePosition(num_units=dim_head, max_relative_position=temporal_length)
self.relative_position_v = RelativePosition(num_units=dim_head, max_relative_position=temporal_length)
self.to_out = nn.Sequential(
nn.Linear(inner_dim, query_dim),
nn.Dropout(dropout)
)
nn.init.constant_(self.to_q.weight, 0)
nn.init.constant_(self.to_k.weight, 0)
nn.init.constant_(self.to_v.weight, 0)
nn.init.constant_(self.to_out[0].weight, 0)
nn.init.constant_(self.to_out[0].bias, 0)
def forward(self, x, context=None, mask=None):
# if context is None:
# print(f'[Temp Attn] x={x.shape},context=None')
# else:
# print(f'[Temp Attn] x={x.shape},context={context.shape}')
nh = self.heads
out = x
q = self.to_q(out)
# if context is not None:
# print(f'temporal context 1 ={context.shape}')
# print(f'x={x.shape}')
context = default(context, x)
# print(f'temporal context 2 ={context.shape}')
k = self.to_k(context)
v = self.to_v(context)
# print(f'q ={q.shape},k={k.shape}')
q, k, v = map(lambda t: rearrange(t, 'b n (h d) -> (b h) n d', h=nh), (q, k, v))
sim = einsum('b i d, b j d -> b i j', q, k) * self.scale
if self.use_relative_position:
len_q, len_k, len_v = q.shape[1], k.shape[1], v.shape[1]
k2 = self.relative_position_k(len_q, len_k)
sim2 = einsum('b t d, t s d -> b t s', q, k2) * self.scale # TODO check
sim += sim2
# print('mask',mask)
if exists(self.mask):
if mask is None:
mask = self.mask.to(sim.device)
else:
mask = self.mask.to(sim.device).bool() & mask #.to(sim.device)
else:
mask = mask
# if self.img_video_joint_train:
# # process mask (make mask same shape with sim)
# c, h, w = mask.shape
# c, t, s = sim.shape
# # assert(h == w and t == s),f"mask={mask.shape}, sim={sim.shape}, h={h}, w={w}, t={t}, s={s}"
# if h > t:
# mask = mask[:, :t, :]
# elif h < t: # pad zeros to mask (no attention) only initial mask =1 area compute weights
# mask_ = torch.zeros([c,t,w]).to(mask.device)
# mask_[:, :h, :] = mask
# mask = mask_
# c, h, w = mask.shape
# if w > s:
# mask = mask[:, :, :s]
# elif w < s: # pad zeros to mask
# mask_ = torch.zeros([c,h,s]).to(mask.device)
# mask_[:, :, :w] = mask
# mask = mask_
# max_neg_value = -torch.finfo(sim.dtype).max
# sim = sim.float().masked_fill(mask == 0, max_neg_value)
if mask is not None:
max_neg_value = -1e9
sim = sim + (1-mask.float()) * max_neg_value # 1=masking,0=no masking
# print('sim after masking: ', sim)
# if torch.isnan(sim).any() or torch.isinf(sim).any() or (not sim.any()):
# print(f'sim [after masking], isnan={torch.isnan(sim).any()}, isinf={torch.isinf(sim).any()}, allzero={not sim.any()}')
attn = sim.softmax(dim=-1)
# print('attn after softmax: ', attn)
# if torch.isnan(attn).any() or torch.isinf(attn).any() or (not attn.any()):
# print(f'attn [after softmax], isnan={torch.isnan(attn).any()}, isinf={torch.isinf(attn).any()}, allzero={not attn.any()}')
# attn = torch.where(torch.isnan(attn), torch.full_like(attn,0), attn)
# if torch.isinf(attn.detach()).any():
# import pdb;pdb.set_trace()
# if torch.isnan(attn.detach()).any():
# import pdb;pdb.set_trace()
out = einsum('b i j, b j d -> b i d', attn, v)
if self.bidirectional_causal_attn:
mask_reverse = torch.triu(torch.ones([1, self.temporal_length, self.temporal_length], device=sim.device))
sim_reverse = sim.float().masked_fill(mask_reverse == 0, max_neg_value)
attn_reverse = sim_reverse.softmax(dim=-1)
out_reverse = einsum('b i j, b j d -> b i d', attn_reverse, v)
out += out_reverse
if self.use_relative_position:
v2 = self.relative_position_v(len_q, len_v)
out2 = einsum('b t s, t s d -> b t d', attn, v2) # TODO check
out += out2 # TODO check:先add还是先merge head?先计算rpr,on split head之后的数据,然后再merge。
out = rearrange(out, '(b h) n d -> b n (h d)', h=nh) # merge head
return self.to_out(out)
class CrossAttention(nn.Module):
def __init__(self, query_dim, context_dim=None, heads=8, dim_head=64, dropout=0.,
sa_shared_kv=False, shared_type='only_first', **kwargs,):
super().__init__()
inner_dim = dim_head * heads
context_dim = default(context_dim, query_dim)
self.sa_shared_kv = sa_shared_kv
assert(shared_type in ['only_first', 'all_frames', 'first_and_prev', 'only_prev', 'full', 'causal', 'full_qkv'])
self.shared_type = shared_type
self.dim_head = dim_head
self.scale = dim_head ** -0.5
self.heads = heads
self.to_q = nn.Linear(query_dim, inner_dim, bias=False)
self.to_k = nn.Linear(context_dim, inner_dim, bias=False)
self.to_v = nn.Linear(context_dim, inner_dim, bias=False)
self.to_out = nn.Sequential(
nn.Linear(inner_dim, query_dim),
nn.Dropout(dropout)
)
if XFORMERS_IS_AVAILBLE:
self.forward = self.efficient_forward
def forward(self, x, context=None, mask=None):
h = self.heads
b = x.shape[0]
q = self.to_q(x)
context = default(context, x)
k = self.to_k(context)
v = self.to_v(context)
if self.sa_shared_kv:
if self.shared_type == 'only_first':
k,v = map(lambda xx: rearrange(xx[0].unsqueeze(0), 'b n c -> (b n) c').unsqueeze(0).repeat(b,1,1),
(k,v))
else:
raise NotImplementedError
q, k, v = map(lambda t: rearrange(t, 'b n (h d) -> (b h) n d', h=h), (q, k, v))
sim = einsum('b i d, b j d -> b i j', q, k) * self.scale
if exists(mask):
mask = rearrange(mask, 'b ... -> b (...)')
max_neg_value = -torch.finfo(sim.dtype).max
mask = repeat(mask, 'b j -> (b h) () j', h=h)
sim.masked_fill_(~mask, max_neg_value)
# attention, what we cannot get enough of
attn = sim.softmax(dim=-1)
out = einsum('b i j, b j d -> b i d', attn, v)
out = rearrange(out, '(b h) n d -> b n (h d)', h=h)
return self.to_out(out)
def efficient_forward(self, x, context=None, mask=None):
q = self.to_q(x)
context = default(context, x)
k = self.to_k(context)
v = self.to_v(context)
b, _, _ = q.shape
q, k, v = map(
lambda t: t.unsqueeze(3)
.reshape(b, t.shape[1], self.heads, self.dim_head)
.permute(0, 2, 1, 3)
.reshape(b * self.heads, t.shape[1], self.dim_head)
.contiguous(),
(q, k, v),
)
# actually compute the attention, what we cannot get enough of
out = xformers.ops.memory_efficient_attention(q, k, v, attn_bias=None, op=None)
if exists(mask):
raise NotImplementedError
out = (
out.unsqueeze(0)
.reshape(b, self.heads, out.shape[1], self.dim_head)
.permute(0, 2, 1, 3)
.reshape(b, out.shape[1], self.heads * self.dim_head)
)
return self.to_out(out)
class VideoSpatialCrossAttention(CrossAttention):
def __init__(self, query_dim, context_dim=None, heads=8, dim_head=64, dropout=0):
super().__init__(query_dim, context_dim, heads, dim_head, dropout)
def forward(self, x, context=None, mask=None):
b, c, t, h, w = x.shape
if context is not None:
context = context.repeat(t, 1, 1)
x = super.forward(spatial_attn_reshape(x), context=context) + x
return spatial_attn_reshape_back(x,b,h)
class BasicTransformerBlockST(nn.Module):
def __init__(self,
# Spatial Stuff
dim,
n_heads,
d_head,
dropout=0.,
context_dim=None,
gated_ff=True,
checkpoint=True,
# Temporal Stuff
temporal_length=None,
image_length=None,
use_relative_position=True,
img_video_joint_train=False,
cross_attn_on_tempoal=False,
temporal_crossattn_type="selfattn",
order="stst",
temporalcrossfirst=False,
temporal_context_dim=None,
split_stcontext=False,
local_spatial_temporal_attn=False,
window_size=2,
**kwargs,
):
super().__init__()
# Self attention
self.attn1 = CrossAttention(query_dim=dim, heads=n_heads, dim_head=d_head, dropout=dropout, **kwargs,)
self.ff = FeedForward(dim, dropout=dropout, glu=gated_ff)
# cross attention if context is not None
self.attn2 = CrossAttention(query_dim=dim, context_dim=context_dim,
heads=n_heads, dim_head=d_head, dropout=dropout, **kwargs,)
self.norm1 = nn.LayerNorm(dim)
self.norm2 = nn.LayerNorm(dim)
self.norm3 = nn.LayerNorm(dim)
self.checkpoint = checkpoint
self.order = order
assert(self.order in ["stst", "sstt", "st_parallel"])
self.temporalcrossfirst = temporalcrossfirst
self.split_stcontext = split_stcontext
self.local_spatial_temporal_attn = local_spatial_temporal_attn
if self.local_spatial_temporal_attn:
assert(self.order == 'stst')
assert(self.order == 'stst')
self.window_size = window_size
if not split_stcontext:
temporal_context_dim = context_dim
# Temporal attention
assert(temporal_crossattn_type in ["selfattn", "crossattn", "skip"])
self.temporal_crossattn_type = temporal_crossattn_type
self.attn1_tmp = TemporalCrossAttention(query_dim=dim, heads=n_heads, dim_head=d_head, dropout=dropout,
temporal_length=temporal_length,
image_length=image_length,
use_relative_position=use_relative_position,
img_video_joint_train=img_video_joint_train,
**kwargs,
)
self.attn2_tmp = TemporalCrossAttention(query_dim=dim, heads=n_heads, dim_head=d_head, dropout=dropout,
# cross attn
context_dim=temporal_context_dim if temporal_crossattn_type == "crossattn" else None,
# temporal attn
temporal_length=temporal_length,
image_length=image_length,
use_relative_position=use_relative_position,
img_video_joint_train=img_video_joint_train,
**kwargs,
)
self.norm4 = nn.LayerNorm(dim)
self.norm5 = nn.LayerNorm(dim)
# self.norm1_tmp = nn.LayerNorm(dim)
# self.norm2_tmp = nn.LayerNorm(dim)
##############################################################################################################################################
def forward(self, x, context=None, temporal_context=None, no_temporal_attn=None, attn_mask=None, **kwargs):
# print(f'no_temporal_attn={no_temporal_attn}')
if not self.split_stcontext:
# st cross attention use the same context vector
temporal_context = context.detach().clone()
if context is None and temporal_context is None:
# self-attention models
if no_temporal_attn:
raise NotImplementedError
return checkpoint(self._forward_nocontext, (x), self.parameters(), self.checkpoint)
else:
# cross-attention models
if no_temporal_attn:
forward_func = self._forward_no_temporal_attn
else:
forward_func = self._forward
inputs = (x, context, temporal_context) if temporal_context is not None else (x, context)
return checkpoint(forward_func, inputs, self.parameters(), self.checkpoint)
# if attn_mask is not None:
# return checkpoint(self._forward, (x, context, temporal_context, attn_mask), self.parameters(), self.checkpoint)
# return checkpoint(self._forward, (x, context, temporal_context), self.parameters(), self.checkpoint)
def _forward(self, x, context=None, temporal_context=None, mask=None, no_temporal_attn=None, ):
assert(x.dim() == 5), f"x shape = {x.shape}"
b, c, t, h, w = x.shape
if self.order in ["stst", "sstt"]:
x = self._st_cross_attn(x, context, temporal_context=temporal_context, order=self.order, mask=mask,)#no_temporal_attn=no_temporal_attn,
elif self.order == "st_parallel":
x = self._st_cross_attn_parallel(x, context, temporal_context=temporal_context, order=self.order,)#no_temporal_attn=no_temporal_attn,
else:
raise NotImplementedError
x = self.ff(self.norm3(x)) + x
if (no_temporal_attn is None) or (not no_temporal_attn):
x = rearrange(x, '(b h w) t c -> b c t h w', b=b,h=h,w=w) # 3d -> 5d
elif no_temporal_attn:
x = rearrange(x, '(b t) (h w) c -> b c t h w', b=b,h=h,w=w) # 3d -> 5d
return x
def _forward_no_temporal_attn(self, x, context=None, temporal_context=None, ):
# temporary implementation :(
# because checkpoint does not support non-tensor inputs currently.
assert(x.dim() == 5), f"x shape = {x.shape}"
b, c, t, h, w = x.shape
if self.order in ["stst", "sstt"]:
# x = self._st_cross_attn(x, context, temporal_context=temporal_context, order=self.order, no_temporal_attn=True,)
# mask = torch.zeros([1, t, t], device=x.device).bool() if context is None else torch.zeros([1, context.shape[1], t], device=x.device).bool()
mask = torch.zeros([1, t, t], device=x.device).bool()
x = self._st_cross_attn(x, context, temporal_context=temporal_context, order=self.order, mask=mask,)
elif self.order == "st_parallel":
x = self._st_cross_attn_parallel(x, context, temporal_context=temporal_context, order=self.order, no_temporal_attn=True,)
else:
raise NotImplementedError
x = self.ff(self.norm3(x)) + x
x = rearrange(x, '(b h w) t c -> b c t h w', b=b,h=h,w=w) # 3d -> 5d
# x = rearrange(x, '(b t) (h w) c -> b c t h w', b=b,h=h,w=w) # 3d -> 5d
return x
def _forward_nocontext(self, x, no_temporal_attn=None):
assert(x.dim() == 5), f"x shape = {x.shape}"
b, c, t, h, w = x.shape
if self.order in ["stst", "sstt"]:
x = self._st_cross_attn(x, order=self.order, no_temporal_attn=no_temporal_attn)
elif self.order == "st_parallel":
x = self._st_cross_attn_parallel(x, order=self.order, no_temporal_attn=no_temporal_attn)
else:
raise NotImplementedError
x = self.ff(self.norm3(x)) + x
x = rearrange(x, '(b h w) t c -> b c t h w', b=b,h=h,w=w) # 3d -> 5d
return x
##############################################################################################################################################
def _st_cross_attn(self, x, context=None, temporal_context=None, order="stst", mask=None): #no_temporal_attn=None,
b, c, t, h, w = x.shape
# print(f'[_st_cross_attn input] x={x.shape}, context={context.shape}')
if order == "stst":
# spatial self attention
x = rearrange(x, 'b c t h w -> (b t) (h w) c')
x = self.attn1(self.norm1(x)) + x
x = rearrange(x, '(b t) (h w) c -> b c t h w', b=b,h=h)
# temporal self attention
# if (no_temporal_attn is None) or (not no_temporal_attn):
if self.local_spatial_temporal_attn:
x = local_spatial_temporal_attn_reshape(x,window_size=self.window_size)
else:
x = rearrange(x, 'b c t h w -> (b h w) t c')
x = self.attn1_tmp(self.norm4(x), mask=mask) + x
if self.local_spatial_temporal_attn:
x = local_spatial_temporal_attn_reshape_back(x, window_size=self.window_size,
b=b, h=h, w=w, t=t)
else:
x = rearrange(x, '(b h w) t c -> b c t h w', b=b,h=h,w=w) # 3d -> 5d
# spatial cross attention
x = rearrange(x, 'b c t h w -> (b t) (h w) c')
# context_ = context.repeat(t, 1, 1) if context is not None else None
# print(f'[before spatial cross] context={context.shape}')
if context is not None:
if context.shape[0] == t: # img captions no_temporal_attn or
context_ = context
else:
context_ = []
for i in range(context.shape[0]):
context_.append(context[i].unsqueeze(0).repeat(t, 1, 1))
context_ = torch.cat(context_,dim=0)
else:
context_ = None
# print(f'[before spatial cross] x={x.shape}, context_={context_.shape}')
x = self.attn2(self.norm2(x), context=context_) + x
# temporal cross attention
# if (no_temporal_attn is None) or (not no_temporal_attn):
x = rearrange(x, '(b t) (h w) c -> b c t h w', b=b,h=h)
x = rearrange(x, 'b c t h w -> (b h w) t c')
if self.temporal_crossattn_type == "crossattn":
# tmporal cross attention
if temporal_context is not None:
# print(f'STATTN context={context.shape}, temporal_context={temporal_context.shape}')
temporal_context = torch.cat([context, temporal_context], dim=1) # blc
# print(f'STATTN after concat temporal_context={temporal_context.shape}')
temporal_context = temporal_context.repeat(h*w, 1,1)
# print(f'after repeat temporal_context={temporal_context.shape}')
else:
temporal_context = context[0:1,...].repeat(h*w, 1, 1)
# print(f'STATTN after concat x={x.shape}')
x = self.attn2_tmp(self.norm5(x), context=temporal_context, mask=mask) + x
elif self.temporal_crossattn_type == "selfattn":
# temporal self attention
x = self.attn2_tmp(self.norm5(x), context=None, mask=mask) + x
elif self.temporal_crossattn_type == "skip":
# no temporal cross and self attention
pass
else:
raise NotImplementedError
elif order == "sstt":
# spatial self attention
x = rearrange(x, 'b c t h w -> (b t) (h w) c')
x = self.attn1(self.norm1(x)) + x
# spatial cross attention
context_ = context.repeat(t, 1, 1) if context is not None else None
x = self.attn2(self.norm2(x), context=context_) + x
x = rearrange(x, '(b t) (h w) c -> b c t h w', b=b,h=h)
if (no_temporal_attn is None) or (not no_temporal_attn):
if self.temporalcrossfirst:
# temporal cross attention
if self.temporal_crossattn_type == "crossattn":
# if temporal_context is not None:
temporal_context = context.repeat(h*w, 1, 1)
x = self.attn2_tmp(self.norm5(x), context=temporal_context, mask=mask) + x
elif self.temporal_crossattn_type == "selfattn":
x = self.attn2_tmp(self.norm5(x), context=None, mask=mask) + x
elif self.temporal_crossattn_type == "skip":
pass
else:
raise NotImplementedError
# temporal self attention
x = rearrange(x, 'b c t h w -> (b h w) t c')
x = self.attn1_tmp(self.norm4(x), mask=mask) + x
else:
# temporal self attention
x = rearrange(x, 'b c t h w -> (b h w) t c')
x = self.attn1_tmp(self.norm4(x), mask=mask) + x
# temporal cross attention
if self.temporal_crossattn_type == "crossattn":
if temporal_context is not None:
temporal_context = context.repeat(h*w, 1, 1)
x = self.attn2_tmp(self.norm5(x), context=temporal_context, mask=mask) + x
elif self.temporal_crossattn_type == "selfattn":
x = self.attn2_tmp(self.norm5(x), context=None, mask=mask) + x
elif self.temporal_crossattn_type == "skip":
pass
else:
raise NotImplementedError
else:
raise NotImplementedError
return x
def _st_cross_attn_parallel(self, x, context=None, temporal_context=None, order="sst", no_temporal_attn=None):
""" order: x -> Self Attn -> Cross Attn -> attn_s
x -> Temp Self Attn -> attn_t
x' = x + attn_s + attn_t
"""
if no_temporal_attn is not None:
raise NotImplementedError
B, C, T, H, W = x.shape
# spatial self attention
h = x
h = rearrange(h, 'b c t h w -> (b t) (h w) c')
h = self.attn1(self.norm1(h)) + h
# spatial cross
# context_ = context.repeat(T, 1, 1) if context is not None else None
if context is not None:
context_ = []
for i in range(context.shape[0]):
context_.append(context[i].unsqueeze(0).repeat(T, 1, 1))
context_ = torch.cat(context_,dim=0)
else:
context_ = None
h = self.attn2(self.norm2(h), context=context_) + h
h = rearrange(h, '(b t) (h w) c -> b c t h w', b=B, h=H)
# temporal self
h2 = x
h2 = rearrange(h2, 'b c t h w -> (b h w) t c')
h2 = self.attn1_tmp(self.norm4(h2))# + h2
h2 = rearrange(h2, '(b h w) t c -> b c t h w', b=B, h=H, w=W)
out = h + h2
return rearrange(out, 'b c t h w -> (b h w) t c')
##############################################################################################################################################
def spatial_attn_reshape(x):
return rearrange(x, 'b c t h w -> (b t) (h w) c')
def spatial_attn_reshape_back(x,b,h):
return rearrange(x, '(b t) (h w) c -> b c t h w', b=b,h=h)
def temporal_attn_reshape(x):
return rearrange(x, 'b c t h w -> (b h w) t c')
def temporal_attn_reshape_back(x, b,h,w):
return rearrange(x, '(b h w) t c -> b c t h w', b=b, h=h, w=w)
def local_spatial_temporal_attn_reshape(x, window_size):
B, C, T, H, W = x.shape
NH = H // window_size
NW = W // window_size
# x = x.view(B, C, T, NH, window_size, NW, window_size)
# tokens = x.permute(0, 1, 2, 3, 5, 4, 6).contiguous()
# tokens = tokens.view(-1, window_size, window_size, C)
x = rearrange(x, 'b c t (nh wh) (nw ww) -> b c t nh wh nw ww', nh=NH, nw=NW, wh=window_size, ww=window_size).contiguous() # # B, C, T, NH, NW, window_size, window_size
x = rearrange(x, 'b c t nh wh nw ww -> (b nh nw) (t wh ww) c') # (B, NH, NW) (T, window_size, window_size) C
return x
def local_spatial_temporal_attn_reshape_back(x, window_size, b, h, w, t):
B, L, C = x.shape
NH = h // window_size
NW = w // window_size
x = rearrange(x, '(b nh nw) (t wh ww) c -> b c t nh wh nw ww', b=b, nh=NH, nw=NW, t=t, wh=window_size, ww=window_size)
x = rearrange(x, 'b c t nh wh nw ww -> b c t (nh wh) (nw ww)')
return x
class SpatialTemporalTransformer(nn.Module):
"""
Transformer block for video-like data (5D tensor).
First, project the input (aka embedding) with NO reshape.
Then apply standard transformer action.
The 5D -> 3D reshape operation will be done in the specific attention module.
"""
def __init__(
self,
in_channels, n_heads, d_head,
depth=1, dropout=0.,
context_dim=None,
# Temporal stuff
temporal_length=None,
image_length=None,
use_relative_position=True,
img_video_joint_train=False,
cross_attn_on_tempoal=False,
temporal_crossattn_type=False,
order="stst",
temporalcrossfirst=False,
split_stcontext=False,
temporal_context_dim=None,
**kwargs,
):
super().__init__()
self.in_channels = in_channels
inner_dim = n_heads * d_head
self.norm = Normalize(in_channels)
self.proj_in = nn.Conv3d(in_channels,
inner_dim,
kernel_size=1,
stride=1,
padding=0)
self.transformer_blocks = nn.ModuleList(
[BasicTransformerBlockST(
inner_dim, n_heads, d_head, dropout=dropout,
# cross attn
context_dim=context_dim,
# temporal attn
temporal_length=temporal_length,
image_length=image_length,
use_relative_position=use_relative_position,
img_video_joint_train=img_video_joint_train,
temporal_crossattn_type=temporal_crossattn_type,
order=order,
temporalcrossfirst=temporalcrossfirst,
split_stcontext=split_stcontext,
temporal_context_dim=temporal_context_dim,
**kwargs
) for d in range(depth)]
)
self.proj_out = zero_module(nn.Conv3d(inner_dim,
in_channels,
kernel_size=1,
stride=1,
padding=0))
def forward(self, x, context=None, temporal_context=None, **kwargs):
# note: if no context is given, cross-attention defaults to self-attention
assert(x.dim() == 5), f"x shape = {x.shape}"
b, c, t, h, w = x.shape
x_in = x
x = self.norm(x)
x = self.proj_in(x)
for block in self.transformer_blocks:
x = block(x, context=context, temporal_context=temporal_context, **kwargs)
x = self.proj_out(x)
return x + x_in
# ---------------------------------------------------------------------------------------------------
class STAttentionBlock2(nn.Module):
def __init__(
self,
channels,
num_heads=1,
num_head_channels=-1,
use_checkpoint=False, # not used, only used in ResBlock
use_new_attention_order=False, # QKVAttention or QKVAttentionLegacy
temporal_length=16, # used in relative positional representation.
image_length=8, # used for image-video joint training.
use_relative_position=False, # whether use relative positional representation in temporal attention.
img_video_joint_train=False,
# norm_type="groupnorm",
attn_norm_type="group",
use_tempoal_causal_attn=False,
):
"""
version 1: guided_diffusion implemented version
version 2: remove args input argument
"""
super().__init__()
if num_head_channels == -1:
self.num_heads = num_heads
else:
assert (
channels % num_head_channels == 0
), f"q,k,v channels {channels} is not divisible by num_head_channels {num_head_channels}"
self.num_heads = channels // num_head_channels
self.use_checkpoint = use_checkpoint
self.temporal_length = temporal_length
self.image_length = image_length
self.use_relative_position = use_relative_position
self.img_video_joint_train = img_video_joint_train
self.attn_norm_type = attn_norm_type
assert(self.attn_norm_type in ["group", "no_norm"])
self.use_tempoal_causal_attn = use_tempoal_causal_attn
if self.attn_norm_type == "group":
self.norm_s = normalization(channels)
self.norm_t = normalization(channels)
self.qkv_s = conv_nd(1, channels, channels * 3, 1)
self.qkv_t = conv_nd(1, channels, channels * 3, 1)
if self.img_video_joint_train:
mask = th.ones([1, temporal_length+image_length, temporal_length+image_length])
mask[:, temporal_length:, :] = 0
mask[:, :, temporal_length:] = 0
self.register_buffer("mask", mask)
else:
self.mask = None
if use_new_attention_order:
# split qkv before split heads
self.attention_s = QKVAttention(self.num_heads)
self.attention_t = QKVAttention(self.num_heads)
else:
# split heads before split qkv
self.attention_s = QKVAttentionLegacy(self.num_heads)
self.attention_t = QKVAttentionLegacy(self.num_heads)
if use_relative_position:
self.relative_position_k = RelativePosition(num_units=channels // self.num_heads, max_relative_position=temporal_length)
self.relative_position_v = RelativePosition(num_units=channels // self.num_heads, max_relative_position=temporal_length)
self.proj_out_s = zero_module(conv_nd(1, channels, channels, 1)) # conv_dim, in_channels, out_channels, kernel_size
self.proj_out_t = zero_module(conv_nd(1, channels, channels, 1)) # conv_dim, in_channels, out_channels, kernel_size
def forward(self, x, mask=None):
b, c, t, h, w = x.shape
# spatial
out = rearrange(x, 'b c t h w -> (b t) c (h w)')
if self.attn_norm_type == "no_norm":
qkv = self.qkv_s(out)
else:
qkv = self.qkv_s(self.norm_s(out))
out = self.attention_s(qkv)
out = self.proj_out_s(out)
out = rearrange(out, '(b t) c (h w) -> b c t h w', b=b,h=h)
x += out
# temporal
out = rearrange(x, 'b c t h w -> (b h w) c t')
if self.attn_norm_type == "no_norm":
qkv = self.qkv_t(out)
else:
qkv = self.qkv_t(self.norm_t(out))
# relative positional embedding
if self.use_relative_position:
len_q = qkv.size()[-1]
len_k, len_v = len_q, len_q
k_rp = self.relative_position_k(len_q, len_k)
v_rp = self.relative_position_v(len_q, len_v) #[T,T,head_dim]
out = self.attention_t(qkv, rp=(k_rp, v_rp), mask=self.mask, use_tempoal_causal_attn=self.use_tempoal_causal_attn)
else:
out = self.attention_t(qkv, rp=None, mask=self.mask, use_tempoal_causal_attn=self.use_tempoal_causal_attn)
out = self.proj_out_t(out)
out = rearrange(out, '(b h w) c t -> b c t h w', b=b,h=h,w=w)
return (x + out)
# ---------------------------------------------------------------------------------------------------------------
class QKVAttentionLegacy(nn.Module):
"""
A module which performs QKV attention. Matches legacy QKVAttention + input/ouput heads shaping
"""
def __init__(self, n_heads):
super().__init__()
self.n_heads = n_heads
def forward(self, qkv, rp=None, mask=None):
"""
Apply QKV attention.
:param qkv: an [N x (H * 3 * C) x T] tensor of Qs, Ks, and Vs.
:return: an [N x (H * C) x T] tensor after attention.
"""
if rp is not None or mask is not None:
raise NotImplementedError
bs, width, length = qkv.shape
assert width % (3 * self.n_heads) == 0
ch = width // (3 * self.n_heads)
q, k, v = qkv.reshape(bs * self.n_heads, ch * 3, length).split(ch, dim=1)
scale = 1 / math.sqrt(math.sqrt(ch))
weight = th.einsum(
"bct,bcs->bts", q * scale, k * scale
) # More stable with f16 than dividing afterwards
weight = th.softmax(weight.float(), dim=-1).type(weight.dtype)
a = th.einsum("bts,bcs->bct", weight, v)
return a.reshape(bs, -1, length)
@staticmethod
def count_flops(model, _x, y):
return count_flops_attn(model, _x, y)
# ---------------------------------------------------------------------------------------------------------------
class QKVAttention(nn.Module):
"""
A module which performs QKV attention and splits in a different order.
"""
def __init__(self, n_heads):
super().__init__()
self.n_heads = n_heads
def forward(self, qkv, rp=None, mask=None, use_tempoal_causal_attn=False):
"""
Apply QKV attention.
:param qkv: an [N x (3 * H * C) x T] tensor of Qs, Ks, and Vs.
:return: an [N x (H * C) x T] tensor after attention.
"""
bs, width, length = qkv.shape
assert width % (3 * self.n_heads) == 0
ch = width // (3 * self.n_heads)
# print('qkv', qkv.size())
q, k, v = qkv.chunk(3, dim=1)
scale = 1 / math.sqrt(math.sqrt(ch))
# print('bs, self.n_heads, ch, length', bs, self.n_heads, ch, length)
weight = th.einsum(
"bct,bcs->bts",
(q * scale).view(bs * self.n_heads, ch, length),
(k * scale).view(bs * self.n_heads, ch, length),
) # More stable with f16 than dividing afterwards
# weight:[b,t,s] b=bs*n_heads*T
if rp is not None:
k_rp, v_rp = rp # [length, length, head_dim] [8, 8, 48]
weight2 = th.einsum(
'bct,tsc->bst',
(q * scale).view(bs * self.n_heads, ch, length),
k_rp
)
weight += weight2
if use_tempoal_causal_attn:
# weight = torch.tril(weight)
assert(mask is None), f'Not implemented for merging two masks!'
mask = torch.tril(torch.ones(weight.shape))
else:
if mask is not None: # only keep upper-left matrix
# process mask
c, t, _ = weight.shape
if mask.shape[-1] > t:
mask = mask[:, :t, :t]
elif mask.shape[-1] < t: # pad ones
mask_ = th.zeros([c,t,t]).to(mask.device)
t_ = mask.shape[-1]
mask_[:, :t_, :t_] = mask
mask = mask_
else:
assert(weight.shape[-1] == mask.shape[-1]), f'weight={weight.shape}, mask={mask.shape}'
if mask is not None:
INF = -1e8 #float('-inf')
weight = weight.float().masked_fill(mask == 0, INF)
weight = F.softmax(weight.float(), dim=-1).type(weight.dtype) #[256, 8, 8] [b, t, t] b=bs*n_heads*h*w,t=nframes
# weight = F.softmax(weight, dim=-1)#[256, 8, 8] [b, t, t] b=bs*n_heads*h*w,t=nframes
a = th.einsum("bts,bcs->bct", weight, v.reshape(bs * self.n_heads, ch, length)) #[256, 48, 8] [b, head_dim, t]
if rp is not None:
a2 = th.einsum(
"bts,tsc->btc",
weight,
v_rp
).transpose(1,2) # btc->bct
a += a2
return a.reshape(bs, -1, length)
# ---------------------------------------------------------------------------------------------------------------
# ---------------------------------------------------------------------------------------------------------------
View File
+102
View File
@@ -0,0 +1,102 @@
import torch.nn as nn
from lvdm.basics import avg_pool_nd, conv_nd
class Downsample(nn.Module):
"""
A downsampling layer with an optional convolution.
:param channels: channels in the inputs and outputs.
:param use_conv: a bool determining if a convolution is applied.
:param dims: determines if the signal is 1D, 2D, or 3D. If 3D, then
downsampling occurs in the inner-two dimensions.
"""
def __init__(self, channels, use_conv, dims=2, out_channels=None, padding=1):
super().__init__()
self.channels = channels
self.out_channels = out_channels or channels
self.use_conv = use_conv
self.dims = dims
stride = 2 if dims != 3 else (1, 2, 2)
if use_conv:
self.op = conv_nd(
dims, self.channels, self.out_channels, 3, stride=stride, padding=padding
)
else:
assert self.channels == self.out_channels
self.op = avg_pool_nd(dims, kernel_size=stride, stride=stride)
def forward(self, x):
assert x.shape[1] == self.channels
return self.op(x)
class ResnetBlock(nn.Module):
def __init__(self, in_c, out_c, down, ksize=3, sk=False, use_conv=True):
super().__init__()
ps = ksize // 2
if in_c != out_c or sk == False:
self.in_conv = nn.Conv2d(in_c, out_c, ksize, 1, ps)
else:
# print('n_in')
self.in_conv = None
self.block1 = nn.Conv2d(out_c, out_c, 3, 1, 1)
self.act = nn.ReLU()
self.block2 = nn.Conv2d(out_c, out_c, ksize, 1, ps)
if sk == False:
# self.skep = nn.Conv2d(in_c, out_c, ksize, 1, ps) # edit by zhouxiawang
self.skep = nn.Conv2d(out_c, out_c, ksize, 1, ps)
else:
self.skep = None
self.down = down
if self.down == True:
self.down_opt = Downsample(in_c, use_conv=use_conv)
def forward(self, x):
if self.down == True:
x = self.down_opt(x)
if self.in_conv is not None: # edit
x = self.in_conv(x)
h = self.block1(x)
h = self.act(h)
h = self.block2(h)
if self.skep is not None:
return h + self.skep(x)
else:
return h + x
class Adapter(nn.Module):
def __init__(self, channels=[320, 640, 1280, 1280], nums_rb=3, cin=64, ksize=3, sk=False, use_conv=True):
super(Adapter, self).__init__()
self.unshuffle = nn.PixelUnshuffle(8)
self.channels = channels
self.nums_rb = nums_rb
self.body = []
for i in range(len(channels)):
for j in range(nums_rb):
if (i != 0) and (j == 0):
self.body.append(
ResnetBlock(channels[i - 1], channels[i], down=True, ksize=ksize, sk=sk, use_conv=use_conv))
else:
self.body.append(
ResnetBlock(channels[i], channels[i], down=False, ksize=ksize, sk=sk, use_conv=use_conv))
self.body = nn.ModuleList(self.body)
self.conv_in = nn.Conv2d(cin, channels[0], 3, 1, 1)
def forward(self, x):
# unshuffle
x = self.unshuffle(x)
# extract features
features = []
x = self.conv_in(x)
for i in range(len(self.channels)):
for j in range(self.nums_rb):
idx = i * self.nums_rb + j
x = self.body[idx](x)
features.append(x)
return features
+352
View File
@@ -0,0 +1,352 @@
import kornia
import open_clip
import torch
import torch.nn as nn
from torch.utils.checkpoint import checkpoint
from transformers import (CLIPTextModel, CLIPTokenizer, T5EncoderModel,
T5Tokenizer)
from lvdm.common import autocast
from utils.utils import count_params
class AbstractEncoder(nn.Module):
def __init__(self):
super().__init__()
def encode(self, *args, **kwargs):
raise NotImplementedError
class IdentityEncoder(AbstractEncoder):
def encode(self, x):
return x
class ClassEmbedder(nn.Module):
def __init__(self, embed_dim, n_classes=1000, key='class', ucg_rate=0.1):
super().__init__()
self.key = key
self.embedding = nn.Embedding(n_classes, embed_dim)
self.n_classes = n_classes
self.ucg_rate = ucg_rate
def forward(self, batch, key=None, disable_dropout=False):
if key is None:
key = self.key
# this is for use in crossattn
c = batch[key][:, None]
if self.ucg_rate > 0. and not disable_dropout:
mask = 1. - torch.bernoulli(torch.ones_like(c) * self.ucg_rate)
c = mask * c + (1 - mask) * torch.ones_like(c) * (self.n_classes - 1)
c = c.long()
c = self.embedding(c)
return c
def get_unconditional_conditioning(self, bs, device="cuda"):
uc_class = self.n_classes - 1 # 1000 classes --> 0 ... 999, one extra class for ucg (class 1000)
uc = torch.ones((bs,), device=device) * uc_class
uc = {self.key: uc}
return uc
def disabled_train(self, mode=True):
"""Overwrite model.train with this function to make sure train/eval mode
does not change anymore."""
return self
class FrozenT5Embedder(AbstractEncoder):
"""Uses the T5 transformer encoder for text"""
def __init__(self, version="google/t5-v1_1-large", device="cuda", max_length=77,
freeze=True): # others are google/t5-v1_1-xl and google/t5-v1_1-xxl
super().__init__()
self.tokenizer = T5Tokenizer.from_pretrained(version)
self.transformer = T5EncoderModel.from_pretrained(version)
self.device = device
self.max_length = max_length # TODO: typical value?
if freeze:
self.freeze()
def freeze(self):
self.transformer = self.transformer.eval()
# self.train = disabled_train
for param in self.parameters():
param.requires_grad = False
def forward(self, text):
batch_encoding = self.tokenizer(text, truncation=True, max_length=self.max_length, return_length=True,
return_overflowing_tokens=False, padding="max_length", return_tensors="pt")
tokens = batch_encoding["input_ids"].to(self.device)
outputs = self.transformer(input_ids=tokens)
z = outputs.last_hidden_state
return z
def encode(self, text):
return self(text)
class FrozenCLIPEmbedder(AbstractEncoder):
"""Uses the CLIP transformer encoder for text (from huggingface)"""
LAYERS = [
"last",
"pooled",
"hidden"
]
def __init__(self, version="openai/clip-vit-large-patch14", device="cuda", max_length=77,
freeze=True, layer="last", layer_idx=None): # clip-vit-base-patch32
super().__init__()
assert layer in self.LAYERS
self.tokenizer = CLIPTokenizer.from_pretrained(version)
self.transformer = CLIPTextModel.from_pretrained(version)
self.device = device
self.max_length = max_length
if freeze:
self.freeze()
self.layer = layer
self.layer_idx = layer_idx
if layer == "hidden":
assert layer_idx is not None
assert 0 <= abs(layer_idx) <= 12
def freeze(self):
self.transformer = self.transformer.eval()
# self.train = disabled_train
for param in self.parameters():
param.requires_grad = False
def forward(self, text):
batch_encoding = self.tokenizer(text, truncation=True, max_length=self.max_length, return_length=True,
return_overflowing_tokens=False, padding="max_length", return_tensors="pt")
tokens = batch_encoding["input_ids"].to(self.device)
outputs = self.transformer(input_ids=tokens, output_hidden_states=self.layer == "hidden")
if self.layer == "last":
z = outputs.last_hidden_state
elif self.layer == "pooled":
z = outputs.pooler_output[:, None, :]
else:
z = outputs.hidden_states[self.layer_idx]
return z
def encode(self, text):
return self(text)
class ClipImageEmbedder(nn.Module):
def __init__(
self,
model,
jit=False,
device='cuda' if torch.cuda.is_available() else 'cpu',
antialias=True,
ucg_rate=0.
):
super().__init__()
from clip import load as load_clip
self.model, _ = load_clip(name=model, device=device, jit=jit)
self.antialias = antialias
self.register_buffer('mean', torch.Tensor([0.48145466, 0.4578275, 0.40821073]), persistent=False)
self.register_buffer('std', torch.Tensor([0.26862954, 0.26130258, 0.27577711]), persistent=False)
self.ucg_rate = ucg_rate
def preprocess(self, x):
# normalize to [0,1]
x = kornia.geometry.resize(x, (224, 224),
interpolation='bicubic', align_corners=True,
antialias=self.antialias)
x = (x + 1.) / 2.
# re-normalize according to clip
x = kornia.enhance.normalize(x, self.mean, self.std)
return x
def forward(self, x, no_dropout=False):
# x is assumed to be in range [-1,1]
out = self.model.encode_image(self.preprocess(x))
out = out.to(x.dtype)
if self.ucg_rate > 0. and not no_dropout:
out = torch.bernoulli((1. - self.ucg_rate) * torch.ones(out.shape[0], device=out.device))[:, None] * out
return out
class FrozenOpenCLIPEmbedder(AbstractEncoder):
"""
Uses the OpenCLIP transformer encoder for text
"""
LAYERS = [
# "pooled",
"last",
"penultimate"
]
def __init__(self, arch="ViT-H-14", version="laion2b_s32b_b79k", device="cuda", max_length=77,
freeze=True, layer="last"):
super().__init__()
assert layer in self.LAYERS
# model, _, _ = open_clip.create_model_and_transforms(arch, device=torch.device('cpu'), pretrained='/apdcephfs/share_1290939/richardxia/PretrainedCache/hub/models--laion--CLIP-ViT-H-14-laion2B-s32B-b79K/snapshots/719803079cc9d41bf3ad0a0916fa24e778320c50/open_clip_pytorch_model.bin')
model, _, _ = open_clip.create_model_and_transforms('hf-hub:laion/CLIP-ViT-H-14-laion2B-s32B-b79K')
del model.visual
self.model = model
self.device = device
self.max_length = max_length
if freeze:
self.freeze()
self.layer = layer
if self.layer == "last":
self.layer_idx = 0
elif self.layer == "penultimate":
self.layer_idx = 1
else:
raise NotImplementedError()
def freeze(self):
self.model = self.model.eval()
for param in self.parameters():
param.requires_grad = False
def forward(self, text):
tokens = open_clip.tokenize(text)
z = self.encode_with_transformer(tokens.to(self.device))
return z
def encode_with_transformer(self, text):
x = self.model.token_embedding(text) # [batch_size, n_ctx, d_model]
x = x + self.model.positional_embedding
x = x.permute(1, 0, 2) # NLD -> LND
x = self.text_transformer_forward(x, attn_mask=self.model.attn_mask)
x = x.permute(1, 0, 2) # LND -> NLD
x = self.model.ln_final(x)
return x
def text_transformer_forward(self, x: torch.Tensor, attn_mask=None):
for i, r in enumerate(self.model.transformer.resblocks):
if i == len(self.model.transformer.resblocks) - self.layer_idx:
break
if self.model.transformer.grad_checkpointing and not torch.jit.is_scripting():
x = checkpoint(r, x, attn_mask)
else:
x = r(x, attn_mask=attn_mask)
return x
def encode(self, text):
return self(text)
class FrozenOpenCLIPImageEmbedder(AbstractEncoder):
"""
Uses the OpenCLIP vision transformer encoder for images
"""
def __init__(self, arch="ViT-H-14", version="laion2b_s32b_b79k", device="cuda", max_length=77,
freeze=True, layer="pooled", antialias=True, ucg_rate=0.):
super().__init__()
model, _, _ = open_clip.create_model_and_transforms(arch, device=torch.device('cpu'),
pretrained=version, )
del model.transformer
self.model = model
self.device = device
self.max_length = max_length
if freeze:
self.freeze()
self.layer = layer
if self.layer == "penultimate":
raise NotImplementedError()
self.layer_idx = 1
self.antialias = antialias
self.register_buffer('mean', torch.Tensor([0.48145466, 0.4578275, 0.40821073]), persistent=False)
self.register_buffer('std', torch.Tensor([0.26862954, 0.26130258, 0.27577711]), persistent=False)
self.ucg_rate = ucg_rate
def preprocess(self, x):
# normalize to [0,1]
x = kornia.geometry.resize(x, (224, 224),
interpolation='bicubic', align_corners=True,
antialias=self.antialias)
x = (x + 1.) / 2.
# renormalize according to clip
x = kornia.enhance.normalize(x, self.mean, self.std)
return x
def freeze(self):
self.model = self.model.eval()
for param in self.parameters():
param.requires_grad = False
@autocast
def forward(self, image, no_dropout=False):
z = self.encode_with_vision_transformer(image)
if self.ucg_rate > 0. and not no_dropout:
z = torch.bernoulli((1. - self.ucg_rate) * torch.ones(z.shape[0], device=z.device))[:, None] * z
return z
def encode_with_vision_transformer(self, img):
img = self.preprocess(img)
x = self.model.visual(img)
return x
def encode(self, text):
return self(text)
class FrozenCLIPT5Encoder(AbstractEncoder):
def __init__(self, clip_version="openai/clip-vit-large-patch14", t5_version="google/t5-v1_1-xl", device="cuda",
clip_max_length=77, t5_max_length=77):
super().__init__()
self.clip_encoder = FrozenCLIPEmbedder(clip_version, device, max_length=clip_max_length)
self.t5_encoder = FrozenT5Embedder(t5_version, device, max_length=t5_max_length)
print(f"{self.clip_encoder.__class__.__name__} has {count_params(self.clip_encoder) * 1.e-6:.2f} M parameters, "
f"{self.t5_encoder.__class__.__name__} comes with {count_params(self.t5_encoder) * 1.e-6:.2f} M params.")
def encode(self, text):
return self(text)
def forward(self, text):
clip_z = self.clip_encoder.encode(text)
t5_z = self.t5_encoder.encode(text)
return [clip_z, t5_z]
'''
from ldm.modules.diffusionmodules.upscaling import ImageConcatWithNoiseAugmentation
from ldm.modules.diffusionmodules.openaimodel import Timestep
class CLIPEmbeddingNoiseAugmentation(ImageConcatWithNoiseAugmentation):
def __init__(self, *args, clip_stats_path=None, timestep_dim=256, **kwargs):
super().__init__(*args, **kwargs)
if clip_stats_path is None:
clip_mean, clip_std = torch.zeros(timestep_dim), torch.ones(timestep_dim)
else:
clip_mean, clip_std = torch.load(clip_stats_path, map_location="cpu")
self.register_buffer("data_mean", clip_mean[None, :], persistent=False)
self.register_buffer("data_std", clip_std[None, :], persistent=False)
self.time_embed = Timestep(timestep_dim)
def scale(self, x):
# re-normalize to centered mean and unit variance
x = (x - self.data_mean) * 1. / self.data_std
return x
def unscale(self, x):
# back to original data stats
x = (x * self.data_std) + self.data_mean
return x
def forward(self, x, noise_level=None):
if noise_level is None:
noise_level = torch.randint(0, self.max_noise_level, (x.shape[0],), device=x.device).long()
else:
assert isinstance(noise_level, torch.Tensor)
x = self.scale(x)
z = self.q_sample(x, noise_level)
z = self.unscale(z)
noise_level = self.time_embed(noise_level)
return z, noise_level
'''
+847
View File
@@ -0,0 +1,847 @@
# pytorch_diffusion + derived encoder decoder
import math
import torch
import numpy as np
import torch.nn as nn
from einops import rearrange
from utils.utils import instantiate_from_config
from lvdm.modules.attention import LinearAttention
def nonlinearity(x):
# swish
return x*torch.sigmoid(x)
def Normalize(in_channels, num_groups=32):
return torch.nn.GroupNorm(num_groups=num_groups, num_channels=in_channels, eps=1e-6, affine=True)
class LinAttnBlock(LinearAttention):
"""to match AttnBlock usage"""
def __init__(self, in_channels):
super().__init__(dim=in_channels, heads=1, dim_head=in_channels)
class AttnBlock(nn.Module):
def __init__(self, in_channels):
super().__init__()
self.in_channels = in_channels
self.norm = Normalize(in_channels)
self.q = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
self.k = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
self.v = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
self.proj_out = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=1,
stride=1,
padding=0)
def forward(self, x):
h_ = x
h_ = self.norm(h_)
q = self.q(h_)
k = self.k(h_)
v = self.v(h_)
# compute attention
b,c,h,w = q.shape
q = q.reshape(b,c,h*w) # bcl
q = q.permute(0,2,1) # bcl -> blc l=hw
k = k.reshape(b,c,h*w) # bcl
w_ = torch.bmm(q,k) # b,hw,hw w[b,i,j]=sum_c q[b,i,c]k[b,c,j]
w_ = w_ * (int(c)**(-0.5))
w_ = torch.nn.functional.softmax(w_, dim=2)
# attend to values
v = v.reshape(b,c,h*w)
w_ = w_.permute(0,2,1) # b,hw,hw (first hw of k, second of q)
h_ = torch.bmm(v,w_) # b, c,hw (hw of q) h_[b,c,j] = sum_i v[b,c,i] w_[b,i,j]
h_ = h_.reshape(b,c,h,w)
h_ = self.proj_out(h_)
return x+h_
def make_attn(in_channels, attn_type="vanilla"):
assert attn_type in ["vanilla", "linear", "none"], f'attn_type {attn_type} unknown'
#print(f"making attention of type '{attn_type}' with {in_channels} in_channels")
if attn_type == "vanilla":
return AttnBlock(in_channels)
elif attn_type == "none":
return nn.Identity(in_channels)
else:
return LinAttnBlock(in_channels)
class Downsample(nn.Module):
def __init__(self, in_channels, with_conv):
super().__init__()
self.with_conv = with_conv
self.in_channels = in_channels
if self.with_conv:
# no asymmetric padding in torch conv, must do it ourselves
self.conv = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=3,
stride=2,
padding=0)
def forward(self, x):
if self.with_conv:
pad = (0,1,0,1)
x = torch.nn.functional.pad(x, pad, mode="constant", value=0)
x = self.conv(x)
else:
x = torch.nn.functional.avg_pool2d(x, kernel_size=2, stride=2)
return x
class Upsample(nn.Module):
def __init__(self, in_channels, with_conv):
super().__init__()
self.with_conv = with_conv
self.in_channels = in_channels
if self.with_conv:
self.conv = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=3,
stride=1,
padding=1)
def forward(self, x):
x = torch.nn.functional.interpolate(x, scale_factor=2.0, mode="nearest")
if self.with_conv:
x = self.conv(x)
return x
def get_timestep_embedding(timesteps, embedding_dim):
"""
This matches the implementation in Denoising Diffusion Probabilistic Models:
From Fairseq.
Build sinusoidal embeddings.
This matches the implementation in tensor2tensor, but differs slightly
from the description in Section 3.5 of "Attention Is All You Need".
"""
assert len(timesteps.shape) == 1
half_dim = embedding_dim // 2
emb = math.log(10000) / (half_dim - 1)
emb = torch.exp(torch.arange(half_dim, dtype=torch.float32) * -emb)
emb = emb.to(device=timesteps.device)
emb = timesteps.float()[:, None] * emb[None, :]
emb = torch.cat([torch.sin(emb), torch.cos(emb)], dim=1)
if embedding_dim % 2 == 1: # zero pad
emb = torch.nn.functional.pad(emb, (0,1,0,0))
return emb
class ResnetBlock(nn.Module):
def __init__(self, *, in_channels, out_channels=None, conv_shortcut=False,
dropout, temb_channels=512):
super().__init__()
self.in_channels = in_channels
out_channels = in_channels if out_channels is None else out_channels
self.out_channels = out_channels
self.use_conv_shortcut = conv_shortcut
self.norm1 = Normalize(in_channels)
self.conv1 = torch.nn.Conv2d(in_channels,
out_channels,
kernel_size=3,
stride=1,
padding=1)
if temb_channels > 0:
self.temb_proj = torch.nn.Linear(temb_channels,
out_channels)
self.norm2 = Normalize(out_channels)
self.dropout = torch.nn.Dropout(dropout)
self.conv2 = torch.nn.Conv2d(out_channels,
out_channels,
kernel_size=3,
stride=1,
padding=1)
if self.in_channels != self.out_channels:
if self.use_conv_shortcut:
self.conv_shortcut = torch.nn.Conv2d(in_channels,
out_channels,
kernel_size=3,
stride=1,
padding=1)
else:
self.nin_shortcut = torch.nn.Conv2d(in_channels,
out_channels,
kernel_size=1,
stride=1,
padding=0)
def forward(self, x, temb):
h = x
h = self.norm1(h)
h = nonlinearity(h)
h = self.conv1(h)
if temb is not None:
h = h + self.temb_proj(nonlinearity(temb))[:,:,None,None]
h = self.norm2(h)
h = nonlinearity(h)
h = self.dropout(h)
h = self.conv2(h)
if self.in_channels != self.out_channels:
if self.use_conv_shortcut:
x = self.conv_shortcut(x)
else:
x = self.nin_shortcut(x)
return x+h
class Model(nn.Module):
def __init__(self, *, ch, out_ch, ch_mult=(1,2,4,8), num_res_blocks,
attn_resolutions, dropout=0.0, resamp_with_conv=True, in_channels,
resolution, use_timestep=True, use_linear_attn=False, attn_type="vanilla"):
super().__init__()
if use_linear_attn: attn_type = "linear"
self.ch = ch
self.temb_ch = self.ch*4
self.num_resolutions = len(ch_mult)
self.num_res_blocks = num_res_blocks
self.resolution = resolution
self.in_channels = in_channels
self.use_timestep = use_timestep
if self.use_timestep:
# timestep embedding
self.temb = nn.Module()
self.temb.dense = nn.ModuleList([
torch.nn.Linear(self.ch,
self.temb_ch),
torch.nn.Linear(self.temb_ch,
self.temb_ch),
])
# downsampling
self.conv_in = torch.nn.Conv2d(in_channels,
self.ch,
kernel_size=3,
stride=1,
padding=1)
curr_res = resolution
in_ch_mult = (1,)+tuple(ch_mult)
self.down = nn.ModuleList()
for i_level in range(self.num_resolutions):
block = nn.ModuleList()
attn = nn.ModuleList()
block_in = ch*in_ch_mult[i_level]
block_out = ch*ch_mult[i_level]
for i_block in range(self.num_res_blocks):
block.append(ResnetBlock(in_channels=block_in,
out_channels=block_out,
temb_channels=self.temb_ch,
dropout=dropout))
block_in = block_out
if curr_res in attn_resolutions:
attn.append(make_attn(block_in, attn_type=attn_type))
down = nn.Module()
down.block = block
down.attn = attn
if i_level != self.num_resolutions-1:
down.downsample = Downsample(block_in, resamp_with_conv)
curr_res = curr_res // 2
self.down.append(down)
# middle
self.mid = nn.Module()
self.mid.block_1 = ResnetBlock(in_channels=block_in,
out_channels=block_in,
temb_channels=self.temb_ch,
dropout=dropout)
self.mid.attn_1 = make_attn(block_in, attn_type=attn_type)
self.mid.block_2 = ResnetBlock(in_channels=block_in,
out_channels=block_in,
temb_channels=self.temb_ch,
dropout=dropout)
# upsampling
self.up = nn.ModuleList()
for i_level in reversed(range(self.num_resolutions)):
block = nn.ModuleList()
attn = nn.ModuleList()
block_out = ch*ch_mult[i_level]
skip_in = ch*ch_mult[i_level]
for i_block in range(self.num_res_blocks+1):
if i_block == self.num_res_blocks:
skip_in = ch*in_ch_mult[i_level]
block.append(ResnetBlock(in_channels=block_in+skip_in,
out_channels=block_out,
temb_channels=self.temb_ch,
dropout=dropout))
block_in = block_out
if curr_res in attn_resolutions:
attn.append(make_attn(block_in, attn_type=attn_type))
up = nn.Module()
up.block = block
up.attn = attn
if i_level != 0:
up.upsample = Upsample(block_in, resamp_with_conv)
curr_res = curr_res * 2
self.up.insert(0, up) # prepend to get consistent order
# end
self.norm_out = Normalize(block_in)
self.conv_out = torch.nn.Conv2d(block_in,
out_ch,
kernel_size=3,
stride=1,
padding=1)
def forward(self, x, t=None, context=None):
#assert x.shape[2] == x.shape[3] == self.resolution
if context is not None:
# assume aligned context, cat along channel axis
x = torch.cat((x, context), dim=1)
if self.use_timestep:
# timestep embedding
assert t is not None
temb = get_timestep_embedding(t, self.ch)
temb = self.temb.dense[0](temb)
temb = nonlinearity(temb)
temb = self.temb.dense[1](temb)
else:
temb = None
# downsampling
hs = [self.conv_in(x)]
for i_level in range(self.num_resolutions):
for i_block in range(self.num_res_blocks):
h = self.down[i_level].block[i_block](hs[-1], temb)
if len(self.down[i_level].attn) > 0:
h = self.down[i_level].attn[i_block](h)
hs.append(h)
if i_level != self.num_resolutions-1:
hs.append(self.down[i_level].downsample(hs[-1]))
# middle
h = hs[-1]
h = self.mid.block_1(h, temb)
h = self.mid.attn_1(h)
h = self.mid.block_2(h, temb)
# upsampling
for i_level in reversed(range(self.num_resolutions)):
for i_block in range(self.num_res_blocks+1):
h = self.up[i_level].block[i_block](
torch.cat([h, hs.pop()], dim=1), temb)
if len(self.up[i_level].attn) > 0:
h = self.up[i_level].attn[i_block](h)
if i_level != 0:
h = self.up[i_level].upsample(h)
# end
h = self.norm_out(h)
h = nonlinearity(h)
h = self.conv_out(h)
return h
def get_last_layer(self):
return self.conv_out.weight
class Encoder(nn.Module):
def __init__(self, *, ch, out_ch, ch_mult=(1,2,4,8), num_res_blocks,
attn_resolutions, dropout=0.0, resamp_with_conv=True, in_channels,
resolution, z_channels, double_z=True, use_linear_attn=False, attn_type="vanilla",
**ignore_kwargs):
super().__init__()
if use_linear_attn: attn_type = "linear"
self.ch = ch
self.temb_ch = 0
self.num_resolutions = len(ch_mult)
self.num_res_blocks = num_res_blocks
self.resolution = resolution
self.in_channels = in_channels
# downsampling
self.conv_in = torch.nn.Conv2d(in_channels,
self.ch,
kernel_size=3,
stride=1,
padding=1)
curr_res = resolution
in_ch_mult = (1,)+tuple(ch_mult)
self.in_ch_mult = in_ch_mult
self.down = nn.ModuleList()
for i_level in range(self.num_resolutions):
block = nn.ModuleList()
attn = nn.ModuleList()
block_in = ch*in_ch_mult[i_level]
block_out = ch*ch_mult[i_level]
for i_block in range(self.num_res_blocks):
block.append(ResnetBlock(in_channels=block_in,
out_channels=block_out,
temb_channels=self.temb_ch,
dropout=dropout))
block_in = block_out
if curr_res in attn_resolutions:
attn.append(make_attn(block_in, attn_type=attn_type))
down = nn.Module()
down.block = block
down.attn = attn
if i_level != self.num_resolutions-1:
down.downsample = Downsample(block_in, resamp_with_conv)
curr_res = curr_res // 2
self.down.append(down)
# middle
self.mid = nn.Module()
self.mid.block_1 = ResnetBlock(in_channels=block_in,
out_channels=block_in,
temb_channels=self.temb_ch,
dropout=dropout)
self.mid.attn_1 = make_attn(block_in, attn_type=attn_type)
self.mid.block_2 = ResnetBlock(in_channels=block_in,
out_channels=block_in,
temb_channels=self.temb_ch,
dropout=dropout)
# end
self.norm_out = Normalize(block_in)
self.conv_out = torch.nn.Conv2d(block_in,
2*z_channels if double_z else z_channels,
kernel_size=3,
stride=1,
padding=1)
def forward(self, x):
# timestep embedding
temb = None
# print(f'encoder-input={x.shape}')
# downsampling
hs = [self.conv_in(x)]
# print(f'encoder-conv in feat={hs[0].shape}')
for i_level in range(self.num_resolutions):
for i_block in range(self.num_res_blocks):
h = self.down[i_level].block[i_block](hs[-1], temb)
# print(f'encoder-down feat={h.shape}')
if len(self.down[i_level].attn) > 0:
h = self.down[i_level].attn[i_block](h)
hs.append(h)
if i_level != self.num_resolutions-1:
# print(f'encoder-downsample (input)={hs[-1].shape}')
hs.append(self.down[i_level].downsample(hs[-1]))
# print(f'encoder-downsample (output)={hs[-1].shape}')
# middle
h = hs[-1]
h = self.mid.block_1(h, temb)
# print(f'encoder-mid1 feat={h.shape}')
h = self.mid.attn_1(h)
h = self.mid.block_2(h, temb)
# print(f'encoder-mid2 feat={h.shape}')
# end
h = self.norm_out(h)
h = nonlinearity(h)
h = self.conv_out(h)
# print(f'end feat={h.shape}')
return h
class Decoder(nn.Module):
def __init__(self, *, ch, out_ch, ch_mult=(1,2,4,8), num_res_blocks,
attn_resolutions, dropout=0.0, resamp_with_conv=True, in_channels,
resolution, z_channels, give_pre_end=False, tanh_out=False, use_linear_attn=False,
attn_type="vanilla", **ignorekwargs):
super().__init__()
if use_linear_attn: attn_type = "linear"
self.ch = ch
self.temb_ch = 0
self.num_resolutions = len(ch_mult)
self.num_res_blocks = num_res_blocks
self.resolution = resolution
self.in_channels = in_channels
self.give_pre_end = give_pre_end
self.tanh_out = tanh_out
# compute in_ch_mult, block_in and curr_res at lowest res
in_ch_mult = (1,)+tuple(ch_mult)
block_in = ch*ch_mult[self.num_resolutions-1]
curr_res = resolution // 2**(self.num_resolutions-1)
self.z_shape = (1,z_channels,curr_res,curr_res)
print("AE working on z of shape {} = {} dimensions.".format(
self.z_shape, np.prod(self.z_shape)))
# z to block_in
self.conv_in = torch.nn.Conv2d(z_channels,
block_in,
kernel_size=3,
stride=1,
padding=1)
# middle
self.mid = nn.Module()
self.mid.block_1 = ResnetBlock(in_channels=block_in,
out_channels=block_in,
temb_channels=self.temb_ch,
dropout=dropout)
self.mid.attn_1 = make_attn(block_in, attn_type=attn_type)
self.mid.block_2 = ResnetBlock(in_channels=block_in,
out_channels=block_in,
temb_channels=self.temb_ch,
dropout=dropout)
# upsampling
self.up = nn.ModuleList()
for i_level in reversed(range(self.num_resolutions)):
block = nn.ModuleList()
attn = nn.ModuleList()
block_out = ch*ch_mult[i_level]
for i_block in range(self.num_res_blocks+1):
block.append(ResnetBlock(in_channels=block_in,
out_channels=block_out,
temb_channels=self.temb_ch,
dropout=dropout))
block_in = block_out
if curr_res in attn_resolutions:
attn.append(make_attn(block_in, attn_type=attn_type))
up = nn.Module()
up.block = block
up.attn = attn
if i_level != 0:
up.upsample = Upsample(block_in, resamp_with_conv)
curr_res = curr_res * 2
self.up.insert(0, up) # prepend to get consistent order
# end
self.norm_out = Normalize(block_in)
self.conv_out = torch.nn.Conv2d(block_in,
out_ch,
kernel_size=3,
stride=1,
padding=1)
def forward(self, z):
#assert z.shape[1:] == self.z_shape[1:]
self.last_z_shape = z.shape
# print(f'decoder-input={z.shape}')
# timestep embedding
temb = None
# z to block_in
h = self.conv_in(z)
# print(f'decoder-conv in feat={h.shape}')
# middle
h = self.mid.block_1(h, temb)
h = self.mid.attn_1(h)
h = self.mid.block_2(h, temb)
# print(f'decoder-mid feat={h.shape}')
# upsampling
for i_level in reversed(range(self.num_resolutions)):
for i_block in range(self.num_res_blocks+1):
h = self.up[i_level].block[i_block](h, temb)
if len(self.up[i_level].attn) > 0:
h = self.up[i_level].attn[i_block](h)
# print(f'decoder-up feat={h.shape}')
if i_level != 0:
h = self.up[i_level].upsample(h)
# print(f'decoder-upsample feat={h.shape}')
# end
if self.give_pre_end:
return h
h = self.norm_out(h)
h = nonlinearity(h)
h = self.conv_out(h)
# print(f'decoder-conv_out feat={h.shape}')
if self.tanh_out:
h = torch.tanh(h)
return h
class SimpleDecoder(nn.Module):
def __init__(self, in_channels, out_channels, *args, **kwargs):
super().__init__()
self.model = nn.ModuleList([nn.Conv2d(in_channels, in_channels, 1),
ResnetBlock(in_channels=in_channels,
out_channels=2 * in_channels,
temb_channels=0, dropout=0.0),
ResnetBlock(in_channels=2 * in_channels,
out_channels=4 * in_channels,
temb_channels=0, dropout=0.0),
ResnetBlock(in_channels=4 * in_channels,
out_channels=2 * in_channels,
temb_channels=0, dropout=0.0),
nn.Conv2d(2*in_channels, in_channels, 1),
Upsample(in_channels, with_conv=True)])
# end
self.norm_out = Normalize(in_channels)
self.conv_out = torch.nn.Conv2d(in_channels,
out_channels,
kernel_size=3,
stride=1,
padding=1)
def forward(self, x):
for i, layer in enumerate(self.model):
if i in [1,2,3]:
x = layer(x, None)
else:
x = layer(x)
h = self.norm_out(x)
h = nonlinearity(h)
x = self.conv_out(h)
return x
class UpsampleDecoder(nn.Module):
def __init__(self, in_channels, out_channels, ch, num_res_blocks, resolution,
ch_mult=(2,2), dropout=0.0):
super().__init__()
# upsampling
self.temb_ch = 0
self.num_resolutions = len(ch_mult)
self.num_res_blocks = num_res_blocks
block_in = in_channels
curr_res = resolution // 2 ** (self.num_resolutions - 1)
self.res_blocks = nn.ModuleList()
self.upsample_blocks = nn.ModuleList()
for i_level in range(self.num_resolutions):
res_block = []
block_out = ch * ch_mult[i_level]
for i_block in range(self.num_res_blocks + 1):
res_block.append(ResnetBlock(in_channels=block_in,
out_channels=block_out,
temb_channels=self.temb_ch,
dropout=dropout))
block_in = block_out
self.res_blocks.append(nn.ModuleList(res_block))
if i_level != self.num_resolutions - 1:
self.upsample_blocks.append(Upsample(block_in, True))
curr_res = curr_res * 2
# end
self.norm_out = Normalize(block_in)
self.conv_out = torch.nn.Conv2d(block_in,
out_channels,
kernel_size=3,
stride=1,
padding=1)
def forward(self, x):
# upsampling
h = x
for k, i_level in enumerate(range(self.num_resolutions)):
for i_block in range(self.num_res_blocks + 1):
h = self.res_blocks[i_level][i_block](h, None)
if i_level != self.num_resolutions - 1:
h = self.upsample_blocks[k](h)
h = self.norm_out(h)
h = nonlinearity(h)
h = self.conv_out(h)
return h
class LatentRescaler(nn.Module):
def __init__(self, factor, in_channels, mid_channels, out_channels, depth=2):
super().__init__()
# residual block, interpolate, residual block
self.factor = factor
self.conv_in = nn.Conv2d(in_channels,
mid_channels,
kernel_size=3,
stride=1,
padding=1)
self.res_block1 = nn.ModuleList([ResnetBlock(in_channels=mid_channels,
out_channels=mid_channels,
temb_channels=0,
dropout=0.0) for _ in range(depth)])
self.attn = AttnBlock(mid_channels)
self.res_block2 = nn.ModuleList([ResnetBlock(in_channels=mid_channels,
out_channels=mid_channels,
temb_channels=0,
dropout=0.0) for _ in range(depth)])
self.conv_out = nn.Conv2d(mid_channels,
out_channels,
kernel_size=1,
)
def forward(self, x):
x = self.conv_in(x)
for block in self.res_block1:
x = block(x, None)
x = torch.nn.functional.interpolate(x, size=(int(round(x.shape[2]*self.factor)), int(round(x.shape[3]*self.factor))))
x = self.attn(x)
for block in self.res_block2:
x = block(x, None)
x = self.conv_out(x)
return x
class MergedRescaleEncoder(nn.Module):
def __init__(self, in_channels, ch, resolution, out_ch, num_res_blocks,
attn_resolutions, dropout=0.0, resamp_with_conv=True,
ch_mult=(1,2,4,8), rescale_factor=1.0, rescale_module_depth=1):
super().__init__()
intermediate_chn = ch * ch_mult[-1]
self.encoder = Encoder(in_channels=in_channels, num_res_blocks=num_res_blocks, ch=ch, ch_mult=ch_mult,
z_channels=intermediate_chn, double_z=False, resolution=resolution,
attn_resolutions=attn_resolutions, dropout=dropout, resamp_with_conv=resamp_with_conv,
out_ch=None)
self.rescaler = LatentRescaler(factor=rescale_factor, in_channels=intermediate_chn,
mid_channels=intermediate_chn, out_channels=out_ch, depth=rescale_module_depth)
def forward(self, x):
x = self.encoder(x)
x = self.rescaler(x)
return x
class MergedRescaleDecoder(nn.Module):
def __init__(self, z_channels, out_ch, resolution, num_res_blocks, attn_resolutions, ch, ch_mult=(1,2,4,8),
dropout=0.0, resamp_with_conv=True, rescale_factor=1.0, rescale_module_depth=1):
super().__init__()
tmp_chn = z_channels*ch_mult[-1]
self.decoder = Decoder(out_ch=out_ch, z_channels=tmp_chn, attn_resolutions=attn_resolutions, dropout=dropout,
resamp_with_conv=resamp_with_conv, in_channels=None, num_res_blocks=num_res_blocks,
ch_mult=ch_mult, resolution=resolution, ch=ch)
self.rescaler = LatentRescaler(factor=rescale_factor, in_channels=z_channels, mid_channels=tmp_chn,
out_channels=tmp_chn, depth=rescale_module_depth)
def forward(self, x):
x = self.rescaler(x)
x = self.decoder(x)
return x
class Upsampler(nn.Module):
def __init__(self, in_size, out_size, in_channels, out_channels, ch_mult=2):
super().__init__()
assert out_size >= in_size
num_blocks = int(np.log2(out_size//in_size))+1
factor_up = 1.+ (out_size % in_size)
print(f"Building {self.__class__.__name__} with in_size: {in_size} --> out_size {out_size} and factor {factor_up}")
self.rescaler = LatentRescaler(factor=factor_up, in_channels=in_channels, mid_channels=2*in_channels,
out_channels=in_channels)
self.decoder = Decoder(out_ch=out_channels, resolution=out_size, z_channels=in_channels, num_res_blocks=2,
attn_resolutions=[], in_channels=None, ch=in_channels,
ch_mult=[ch_mult for _ in range(num_blocks)])
def forward(self, x):
x = self.rescaler(x)
x = self.decoder(x)
return x
class Resize(nn.Module):
def __init__(self, in_channels=None, learned=False, mode="bilinear"):
super().__init__()
self.with_conv = learned
self.mode = mode
if self.with_conv:
print(f"Note: {self.__class__.__name} uses learned downsampling and will ignore the fixed {mode} mode")
raise NotImplementedError()
assert in_channels is not None
# no asymmetric padding in torch conv, must do it ourselves
self.conv = torch.nn.Conv2d(in_channels,
in_channels,
kernel_size=4,
stride=2,
padding=1)
def forward(self, x, scale_factor=1.0):
if scale_factor==1.0:
return x
else:
x = torch.nn.functional.interpolate(x, mode=self.mode, align_corners=False, scale_factor=scale_factor)
return x
class FirstStagePostProcessor(nn.Module):
def __init__(self, ch_mult:list, in_channels,
pretrained_model:nn.Module=None,
reshape=False,
n_channels=None,
dropout=0.,
pretrained_config=None):
super().__init__()
if pretrained_config is None:
assert pretrained_model is not None, 'Either "pretrained_model" or "pretrained_config" must not be None'
self.pretrained_model = pretrained_model
else:
assert pretrained_config is not None, 'Either "pretrained_model" or "pretrained_config" must not be None'
self.instantiate_pretrained(pretrained_config)
self.do_reshape = reshape
if n_channels is None:
n_channels = self.pretrained_model.encoder.ch
self.proj_norm = Normalize(in_channels,num_groups=in_channels//2)
self.proj = nn.Conv2d(in_channels,n_channels,kernel_size=3,
stride=1,padding=1)
blocks = []
downs = []
ch_in = n_channels
for m in ch_mult:
blocks.append(ResnetBlock(in_channels=ch_in,out_channels=m*n_channels,dropout=dropout))
ch_in = m * n_channels
downs.append(Downsample(ch_in, with_conv=False))
self.model = nn.ModuleList(blocks)
self.downsampler = nn.ModuleList(downs)
def instantiate_pretrained(self, config):
model = instantiate_from_config(config)
self.pretrained_model = model.eval()
# self.pretrained_model.train = False
for param in self.pretrained_model.parameters():
param.requires_grad = False
@torch.no_grad()
def encode_with_pretrained(self,x):
c = self.pretrained_model.encode(x)
if isinstance(c, DiagonalGaussianDistribution):
c = c.mode()
return c
def forward(self,x):
z_fs = self.encode_with_pretrained(x)
z = self.proj_norm(z_fs)
z = self.proj(z)
z = nonlinearity(z)
for submodel, downmodel in zip(self.model,self.downsampler):
z = submodel(z,temb=None)
z = downmodel(z)
if self.do_reshape:
z = rearrange(z,'b c h w -> b (h w) c')
return z
+580
View File
@@ -0,0 +1,580 @@
import math
import random
from abc import abstractmethod
from functools import partial
import numpy as np
import torch
import torch.nn as nn
import torch.nn.functional as F
from einops import rearrange, repeat
from lvdm.basics import (avg_pool_nd, conv_nd, linear, normalization,
zero_module)
from lvdm.common import checkpoint
from lvdm.models.utils_diffusion import timestep_embedding
from lvdm.modules.attention import SpatialTransformer, TemporalTransformer
class TimestepBlock(nn.Module):
"""
Any module where forward() takes timestep embeddings as a second argument.
"""
@abstractmethod
def forward(self, x, emb):
"""
Apply the module to `x` given `emb` timestep embeddings.
"""
class TimestepEmbedSequential(nn.Sequential, TimestepBlock):
"""
A sequential module that passes timestep embeddings to the children that
support it as an extra input.
"""
def forward(self, x, emb, context=None, batch_size=None, is_imgbatch=False):
for layer in self:
if isinstance(layer, TimestepBlock):
x = layer(x, emb, batch_size, is_imgbatch=is_imgbatch)
elif isinstance(layer, SpatialTransformer):
x = layer(x, context)
elif isinstance(layer, TemporalTransformer):
x = rearrange(x, '(b f) c h w -> b c f h w', b=batch_size)
x = layer(x, context, is_imgbatch=is_imgbatch)
x = rearrange(x, 'b c f h w -> (b f) c h w')
else:
x = layer(x,)
return x
class Downsample(nn.Module):
"""
A downsampling layer with an optional convolution.
:param channels: channels in the inputs and outputs.
:param use_conv: a bool determining if a convolution is applied.
:param dims: determines if the signal is 1D, 2D, or 3D. If 3D, then
downsampling occurs in the inner-two dimensions.
"""
def __init__(self, channels, use_conv, dims=2, out_channels=None, padding=1):
super().__init__()
self.channels = channels
self.out_channels = out_channels or channels
self.use_conv = use_conv
self.dims = dims
stride = 2 if dims != 3 else (1, 2, 2)
if use_conv:
self.op = conv_nd(
dims, self.channels, self.out_channels, 3, stride=stride, padding=padding
)
else:
assert self.channels == self.out_channels
self.op = avg_pool_nd(dims, kernel_size=stride, stride=stride)
def forward(self, x):
assert x.shape[1] == self.channels
return self.op(x)
class Upsample(nn.Module):
"""
An upsampling layer with an optional convolution.
:param channels: channels in the inputs and outputs.
:param use_conv: a bool determining if a convolution is applied.
:param dims: determines if the signal is 1D, 2D, or 3D. If 3D, then
upsampling occurs in the inner-two dimensions.
"""
def __init__(self, channels, use_conv, dims=2, out_channels=None, padding=1):
super().__init__()
self.channels = channels
self.out_channels = out_channels or channels
self.use_conv = use_conv
self.dims = dims
if use_conv:
self.conv = conv_nd(dims, self.channels, self.out_channels, 3, padding=padding)
def forward(self, x):
assert x.shape[1] == self.channels
if self.dims == 3:
x = F.interpolate(x, (x.shape[2], x.shape[3] * 2, x.shape[4] * 2), mode='nearest')
else:
x = F.interpolate(x, scale_factor=2, mode='nearest')
if self.use_conv:
x = self.conv(x)
return x
class ResBlock(TimestepBlock):
"""
A residual block that can optionally change the number of channels.
:param channels: the number of input channels.
:param emb_channels: the number of timestep embedding channels.
:param dropout: the rate of dropout.
:param out_channels: if specified, the number of out channels.
:param use_conv: if True and out_channels is specified, use a spatial
convolution instead of a smaller 1x1 convolution to change the
channels in the skip connection.
:param dims: determines if the signal is 1D, 2D, or 3D.
:param up: if True, use this block for upsampling.
:param down: if True, use this block for downsampling.
:param use_temporal_conv: if True, use the temporal convolution.
:param use_image_dataset: if True, the temporal parameters will not be optimized.
"""
def __init__(
self,
channels,
emb_channels,
dropout,
out_channels=None,
use_scale_shift_norm=False,
dims=2,
use_checkpoint=False,
use_conv=False,
up=False,
down=False,
use_temporal_conv=False,
tempspatial_aware=False,
use_image_dataset=False,
):
super().__init__()
self.channels = channels
self.emb_channels = emb_channels
self.dropout = dropout
self.out_channels = out_channels or channels
self.use_conv = use_conv
self.use_checkpoint = use_checkpoint
self.use_scale_shift_norm = use_scale_shift_norm
self.use_temporal_conv = use_temporal_conv
self.in_layers = nn.Sequential(
normalization(channels),
nn.SiLU(),
conv_nd(dims, channels, self.out_channels, 3, padding=1),
)
self.updown = up or down
if up:
self.h_upd = Upsample(channels, False, dims)
self.x_upd = Upsample(channels, False, dims)
elif down:
self.h_upd = Downsample(channels, False, dims)
self.x_upd = Downsample(channels, False, dims)
else:
self.h_upd = self.x_upd = nn.Identity()
self.emb_layers = nn.Sequential(
nn.SiLU(),
nn.Linear(
emb_channels,
2 * self.out_channels if use_scale_shift_norm else self.out_channels,
),
)
self.out_layers = nn.Sequential(
normalization(self.out_channels),
nn.SiLU(),
nn.Dropout(p=dropout),
zero_module(nn.Conv2d(self.out_channels, self.out_channels, 3, padding=1)),
)
if self.out_channels == channels:
self.skip_connection = nn.Identity()
elif use_conv:
self.skip_connection = conv_nd(dims, channels, self.out_channels, 3, padding=1)
else:
self.skip_connection = conv_nd(dims, channels, self.out_channels, 1)
if self.use_temporal_conv:
self.temopral_conv = TemporalConvBlock(
self.out_channels,
self.out_channels,
dropout=0.1,
spatial_aware=tempspatial_aware,
use_image_dataset=use_image_dataset
)
def forward(self, x, emb, batch_size=None, is_imgbatch=False):
"""
Apply the block to a Tensor, conditioned on a timestep embedding.
:param x: an [N x C x ...] Tensor of features.
:param emb: an [N x emb_channels] Tensor of timestep embeddings.
:return: an [N x C x ...] Tensor of outputs.
"""
input_tuple = (x, emb)
if self.use_temporal_conv:
forward_tempconv = partial(self._forward, batch_size=batch_size, is_imgbatch=is_imgbatch)
return checkpoint(forward_tempconv, input_tuple, self.parameters(), self.use_checkpoint)
return checkpoint(self._forward, input_tuple, self.parameters(), self.use_checkpoint)
def _forward(self, x, emb, batch_size=None, is_imgbatch=False):
if self.updown:
in_rest, in_conv = self.in_layers[:-1], self.in_layers[-1]
h = in_rest(x)
h = self.h_upd(h)
x = self.x_upd(x)
h = in_conv(h)
else:
h = self.in_layers(x)
emb_out = self.emb_layers(emb).type(h.dtype)
while len(emb_out.shape) < len(h.shape):
emb_out = emb_out[..., None]
if self.use_scale_shift_norm:
out_norm, out_rest = self.out_layers[0], self.out_layers[1:]
scale, shift = torch.chunk(emb_out, 2, dim=1)
h = out_norm(h) * (1 + scale) + shift
h = out_rest(h)
else:
h = h + emb_out
h = self.out_layers(h)
h = self.skip_connection(x) + h
if self.use_temporal_conv and batch_size and not is_imgbatch:
h = rearrange(h, '(b t) c h w -> b c t h w', b=batch_size)
h = self.temopral_conv(h)
h = rearrange(h, 'b c t h w -> (b t) c h w')
return h
class TemporalConvBlock(nn.Module):
def __init__(self, in_channels, out_channels=None, dropout=0.0, spatial_aware=False, use_image_dataset=False):
super(TemporalConvBlock, self).__init__()
if out_channels is None:
out_channels = in_channels # int(1.5*in_channels)
self.in_channels = in_channels
self.out_channels = out_channels
self.use_image_dataset = use_image_dataset
kernel_shape = (3, 1, 1) if not spatial_aware else (3, 3, 3)
padding_shape = (1, 0, 0) if not spatial_aware else (1, 1, 1)
# conv layers
self.conv1 = nn.Sequential(
nn.GroupNorm(32, in_channels), nn.SiLU(),
nn.Conv3d(in_channels, out_channels, kernel_shape, padding=padding_shape))
self.conv2 = nn.Sequential(
nn.GroupNorm(32, out_channels), nn.SiLU(), nn.Dropout(dropout),
nn.Conv3d(out_channels, in_channels, kernel_shape, padding=padding_shape))
self.conv3 = nn.Sequential(
nn.GroupNorm(32, out_channels), nn.SiLU(), nn.Dropout(dropout),
nn.Conv3d(out_channels, in_channels, (3, 1, 1), padding=(1, 0, 0)))
self.conv4 = nn.Sequential(
nn.GroupNorm(32, out_channels), nn.SiLU(), nn.Dropout(dropout),
nn.Conv3d(out_channels, in_channels, (3, 1, 1), padding=(1, 0, 0)))
# zero out the last layer params,so the conv block is identity
nn.init.zeros_(self.conv4[-1].weight)
nn.init.zeros_(self.conv4[-1].bias)
def forward(self, x):
identity = x
x = self.conv1(x)
x = self.conv2(x)
x = self.conv3(x)
x = self.conv4(x)
if self.use_image_dataset:
x = identity + 0.0 * x
else:
x = identity + x
return x
class UNetModel(nn.Module):
"""
The full UNet model with attention and timestep embedding.
:param in_channels: in_channels in the input Tensor.
:param model_channels: base channel count for the model.
:param out_channels: channels in the output Tensor.
:param num_res_blocks: number of residual blocks per downsample.
:param attention_resolutions: a collection of downsample rates at which
attention will take place. May be a set, list, or tuple.
For example, if this contains 4, then at 4x downsampling, attention
will be used.
:param dropout: the dropout probability.
:param channel_mult: channel multiplier for each level of the UNet.
:param conv_resample: if True, use learned convolutions for upsampling and
downsampling.
:param dims: determines if the signal is 1D, 2D, or 3D.
:param num_classes: if specified (as an int), then this model will be
class-conditional with `num_classes` classes.
:param use_checkpoint: use gradient checkpointing to reduce memory usage.
:param num_heads: the number of attention heads in each attention layer.
:param num_heads_channels: if specified, ignore num_heads and instead use
a fixed channel width per attention head.
:param num_heads_upsample: works with num_heads to set a different number
of heads for upsampling. Deprecated.
:param use_scale_shift_norm: use a FiLM-like conditioning mechanism.
:param resblock_updown: use residual blocks for up/downsampling.
:param use_new_attention_order: use a different attention pattern for potentially
increased efficiency.
"""
def __init__(self,
in_channels,
model_channels,
out_channels,
num_res_blocks,
attention_resolutions,
dropout=0.0,
channel_mult=(1, 2, 4, 8),
conv_resample=True,
dims=2,
context_dim=None,
use_scale_shift_norm=False,
resblock_updown=False,
num_heads=-1,
num_head_channels=-1,
transformer_depth=1,
use_linear=False,
use_checkpoint=False,
temporal_conv=False,
tempspatial_aware=False,
temporal_attention=True,
addition_attention=False,
temporal_selfatt_only=True,
use_relative_position=True,
use_causal_attention=False,
temporal_length=None,
use_image_dataset=False,
use_fp16=False,
micro_condition=False,
temporal_transformer_depth=1
):
super(UNetModel, self).__init__()
if num_heads == -1:
assert num_head_channels != -1, 'Either num_heads or num_head_channels has to be set'
if num_head_channels == -1:
assert num_heads != -1, 'Either num_heads or num_head_channels has to be set'
self.in_channels = in_channels
self.model_channels = model_channels
self.out_channels = out_channels
self.num_res_blocks = num_res_blocks
self.attention_resolutions = attention_resolutions
self.dropout = dropout
self.channel_mult = channel_mult
self.conv_resample = conv_resample
self.temporal_attention = temporal_attention
time_embed_dim = model_channels * 4
self.use_checkpoint = use_checkpoint
self.dtype = torch.float16 if use_fp16 else torch.float32
#temporal_selfatt_only = True
self.addition_attention=addition_attention
self.time_embed = nn.Sequential(
linear(model_channels, time_embed_dim),
nn.SiLU(),
linear(time_embed_dim, time_embed_dim),
)
if micro_condition:
self.micro_embed = nn.Sequential(
linear(model_channels, time_embed_dim),
nn.SiLU(),
linear(time_embed_dim, time_embed_dim),
)
self.micro_condition = micro_condition
self.input_blocks = nn.ModuleList(
[
TimestepEmbedSequential(conv_nd(dims, in_channels, model_channels, 3, padding=1))
]
)
if self.addition_attention:
self.init_attn=TimestepEmbedSequential(
TemporalTransformer(
model_channels,
n_heads=8,
d_head=num_head_channels,
depth=transformer_depth,
context_dim=context_dim,
use_checkpoint=use_checkpoint, only_self_att=temporal_selfatt_only,
causal_attention=use_causal_attention, relative_position=use_relative_position,
temporal_length=temporal_length, use_image_dataset=use_image_dataset))
input_block_chans = [model_channels]
ch = model_channels
ds = 1
for level, mult in enumerate(channel_mult):
for _ in range(num_res_blocks):
layers = [
ResBlock(ch, time_embed_dim, dropout,
out_channels=mult * model_channels, dims=dims, use_checkpoint=use_checkpoint,
use_scale_shift_norm=use_scale_shift_norm, tempspatial_aware=tempspatial_aware,
use_temporal_conv=temporal_conv, use_image_dataset=use_image_dataset
)
]
ch = mult * model_channels
if ds in attention_resolutions:
if num_head_channels == -1:
dim_head = ch // num_heads
else:
num_heads = ch // num_head_channels
dim_head = num_head_channels
layers.append(
SpatialTransformer(ch, num_heads, dim_head,
depth=transformer_depth, context_dim=context_dim, use_linear=use_linear,
use_checkpoint=use_checkpoint, disable_self_attn=False
)
)
if self.temporal_attention:
layers.append(
TemporalTransformer(ch, num_heads, dim_head,
depth=temporal_transformer_depth, context_dim=context_dim, use_linear=use_linear,
use_checkpoint=use_checkpoint, only_self_att=temporal_selfatt_only,
causal_attention=use_causal_attention, relative_position=use_relative_position,
temporal_length=temporal_length, use_image_dataset=use_image_dataset
)
)
self.input_blocks.append(TimestepEmbedSequential(*layers))
input_block_chans.append(ch)
if level != len(channel_mult) - 1:
out_ch = ch
self.input_blocks.append(
TimestepEmbedSequential(
ResBlock(ch, time_embed_dim, dropout,
out_channels=out_ch, dims=dims, use_checkpoint=use_checkpoint,
use_scale_shift_norm=use_scale_shift_norm,
down=True
)
if resblock_updown
else Downsample(ch, conv_resample, dims=dims, out_channels=out_ch)
)
)
ch = out_ch
input_block_chans.append(ch)
ds *= 2
if num_head_channels == -1:
dim_head = ch // num_heads
else:
num_heads = ch // num_head_channels
dim_head = num_head_channels
layers = [
ResBlock(ch, time_embed_dim, dropout,
dims=dims, use_checkpoint=use_checkpoint,
use_scale_shift_norm=use_scale_shift_norm, tempspatial_aware=tempspatial_aware,
use_temporal_conv=temporal_conv, use_image_dataset=use_image_dataset
),
SpatialTransformer(ch, num_heads, dim_head,
depth=transformer_depth, context_dim=context_dim, use_linear=use_linear,
use_checkpoint=use_checkpoint, disable_self_attn=False
)
]
if self.temporal_attention:
layers.append(
TemporalTransformer(ch, num_heads, dim_head,
depth=temporal_transformer_depth, context_dim=context_dim, use_linear=use_linear,
use_checkpoint=use_checkpoint, only_self_att=temporal_selfatt_only,
causal_attention=use_causal_attention, relative_position=use_relative_position,
temporal_length=temporal_length, use_image_dataset=use_image_dataset
)
)
layers.append(
ResBlock(ch, time_embed_dim, dropout,
dims=dims, use_checkpoint=use_checkpoint,
use_scale_shift_norm=use_scale_shift_norm, tempspatial_aware=tempspatial_aware,
use_temporal_conv=temporal_conv, use_image_dataset=use_image_dataset
)
)
self.middle_block = TimestepEmbedSequential(*layers)
self.output_blocks = nn.ModuleList([])
for level, mult in list(enumerate(channel_mult))[::-1]:
for i in range(num_res_blocks + 1):
ich = input_block_chans.pop()
layers = [
ResBlock(ch + ich, time_embed_dim, dropout,
out_channels=mult * model_channels, dims=dims, use_checkpoint=use_checkpoint,
use_scale_shift_norm=use_scale_shift_norm, tempspatial_aware=tempspatial_aware,
use_temporal_conv=temporal_conv, use_image_dataset=use_image_dataset
)
]
ch = model_channels * mult
if ds in attention_resolutions:
if num_head_channels == -1:
dim_head = ch // num_heads
else:
num_heads = ch // num_head_channels
dim_head = num_head_channels
layers.append(
SpatialTransformer(ch, num_heads, dim_head,
depth=transformer_depth, context_dim=context_dim, use_linear=use_linear,
use_checkpoint=use_checkpoint, disable_self_attn=False
)
)
if self.temporal_attention:
layers.append(
TemporalTransformer(ch, num_heads, dim_head,
depth=temporal_transformer_depth, context_dim=context_dim, use_linear=use_linear,
use_checkpoint=use_checkpoint, only_self_att=temporal_selfatt_only,
causal_attention=use_causal_attention, relative_position=use_relative_position,
temporal_length=temporal_length, use_image_dataset=use_image_dataset
)
)
if level and i == num_res_blocks:
out_ch = ch
layers.append(
ResBlock(ch, time_embed_dim, dropout,
out_channels=out_ch, dims=dims, use_checkpoint=use_checkpoint,
use_scale_shift_norm=use_scale_shift_norm,
up=True
)
if resblock_updown
else Upsample(ch, conv_resample, dims=dims, out_channels=out_ch)
)
ds //= 2
self.output_blocks.append(TimestepEmbedSequential(*layers))
self.out = nn.Sequential(
normalization(ch),
nn.SiLU(),
zero_module(conv_nd(dims, model_channels, out_channels, 3, padding=1)),
)
def forward(self, x, timesteps, context=None, y=None, features_adapter=None, is_imgbatch=False, **kwargs):
b,_,t,_,_ = x.shape
t_emb = timestep_embedding(timesteps, self.model_channels, repeat_only=False)
emb = self.time_embed(t_emb)
if self.micro_condition and y is not None:
micro_emb = timestep_embedding(y, self.model_channels, repeat_only=False)
emb = emb + self.micro_embed(micro_emb)
## repeat t times for context [(b t) 77 768] & time embedding
if not is_imgbatch:
context = context.repeat_interleave(repeats=t, dim=0)
emb = emb.repeat_interleave(repeats=t, dim=0)
## always in shape (b t) c h w, except for temporal layer
x = rearrange(x, 'b c t h w -> (b t) c h w')
if features_adapter is not None:
features_adapter = [rearrange(feature, 'b c t h w -> (b t) c h w') for feature in features_adapter]
h = x.type(self.dtype)
adapter_idx = 0
hs = []
for id, module in enumerate(self.input_blocks):
h = module(h, emb, context=context, batch_size=b,is_imgbatch=is_imgbatch)
if id ==0 and self.addition_attention:
h = self.init_attn(h, emb, context=context, batch_size=b,is_imgbatch=is_imgbatch)
## plug-in adapter features
if ((id+1)%3 == 0) and features_adapter is not None:
h = h + features_adapter[adapter_idx]
adapter_idx += 1
hs.append(h)
if features_adapter is not None:
assert len(features_adapter)==adapter_idx, 'Wrong features_adapter'
h = self.middle_block(h, emb, context=context, batch_size=b, is_imgbatch=is_imgbatch)
for module in self.output_blocks:
h = torch.cat([h, hs.pop()], dim=1)
h = module(h, emb, context=context, batch_size=b, is_imgbatch=is_imgbatch)
h = h.type(x.dtype)
y = self.out(h)
# reshape back to (b c t h w)
y = rearrange(y, '(b t) c h w -> b c t h w', b=b)
return y
+641
View File
@@ -0,0 +1,641 @@
"""shout-out to https://github.com/lucidrains/x-transformers/tree/main/x_transformers"""
from functools import partial
from inspect import isfunction
from collections import namedtuple
from einops import rearrange, repeat
import torch
from torch import nn, einsum
import torch.nn.functional as F
# constants
DEFAULT_DIM_HEAD = 64
Intermediates = namedtuple('Intermediates', [
'pre_softmax_attn',
'post_softmax_attn'
])
LayerIntermediates = namedtuple('Intermediates', [
'hiddens',
'attn_intermediates'
])
class AbsolutePositionalEmbedding(nn.Module):
def __init__(self, dim, max_seq_len):
super().__init__()
self.emb = nn.Embedding(max_seq_len, dim)
self.init_()
def init_(self):
nn.init.normal_(self.emb.weight, std=0.02)
def forward(self, x):
n = torch.arange(x.shape[1], device=x.device)
return self.emb(n)[None, :, :]
class FixedPositionalEmbedding(nn.Module):
def __init__(self, dim):
super().__init__()
inv_freq = 1. / (10000 ** (torch.arange(0, dim, 2).float() / dim))
self.register_buffer('inv_freq', inv_freq)
def forward(self, x, seq_dim=1, offset=0):
t = torch.arange(x.shape[seq_dim], device=x.device).type_as(self.inv_freq) + offset
sinusoid_inp = torch.einsum('i , j -> i j', t, self.inv_freq)
emb = torch.cat((sinusoid_inp.sin(), sinusoid_inp.cos()), dim=-1)
return emb[None, :, :]
# helpers
def exists(val):
return val is not None
def default(val, d):
if exists(val):
return val
return d() if isfunction(d) else d
def always(val):
def inner(*args, **kwargs):
return val
return inner
def not_equals(val):
def inner(x):
return x != val
return inner
def equals(val):
def inner(x):
return x == val
return inner
def max_neg_value(tensor):
return -torch.finfo(tensor.dtype).max
# keyword argument helpers
def pick_and_pop(keys, d):
values = list(map(lambda key: d.pop(key), keys))
return dict(zip(keys, values))
def group_dict_by_key(cond, d):
return_val = [dict(), dict()]
for key in d.keys():
match = bool(cond(key))
ind = int(not match)
return_val[ind][key] = d[key]
return (*return_val,)
def string_begins_with(prefix, str):
return str.startswith(prefix)
def group_by_key_prefix(prefix, d):
return group_dict_by_key(partial(string_begins_with, prefix), d)
def groupby_prefix_and_trim(prefix, d):
kwargs_with_prefix, kwargs = group_dict_by_key(partial(string_begins_with, prefix), d)
kwargs_without_prefix = dict(map(lambda x: (x[0][len(prefix):], x[1]), tuple(kwargs_with_prefix.items())))
return kwargs_without_prefix, kwargs
# classes
class Scale(nn.Module):
def __init__(self, value, fn):
super().__init__()
self.value = value
self.fn = fn
def forward(self, x, **kwargs):
x, *rest = self.fn(x, **kwargs)
return (x * self.value, *rest)
class Rezero(nn.Module):
def __init__(self, fn):
super().__init__()
self.fn = fn
self.g = nn.Parameter(torch.zeros(1))
def forward(self, x, **kwargs):
x, *rest = self.fn(x, **kwargs)
return (x * self.g, *rest)
class ScaleNorm(nn.Module):
def __init__(self, dim, eps=1e-5):
super().__init__()
self.scale = dim ** -0.5
self.eps = eps
self.g = nn.Parameter(torch.ones(1))
def forward(self, x):
norm = torch.norm(x, dim=-1, keepdim=True) * self.scale
return x / norm.clamp(min=self.eps) * self.g
class RMSNorm(nn.Module):
def __init__(self, dim, eps=1e-8):
super().__init__()
self.scale = dim ** -0.5
self.eps = eps
self.g = nn.Parameter(torch.ones(dim))
def forward(self, x):
norm = torch.norm(x, dim=-1, keepdim=True) * self.scale
return x / norm.clamp(min=self.eps) * self.g
class Residual(nn.Module):
def forward(self, x, residual):
return x + residual
class GRUGating(nn.Module):
def __init__(self, dim):
super().__init__()
self.gru = nn.GRUCell(dim, dim)
def forward(self, x, residual):
gated_output = self.gru(
rearrange(x, 'b n d -> (b n) d'),
rearrange(residual, 'b n d -> (b n) d')
)
return gated_output.reshape_as(x)
# feedforward
class GEGLU(nn.Module):
def __init__(self, dim_in, dim_out):
super().__init__()
self.proj = nn.Linear(dim_in, dim_out * 2)
def forward(self, x):
x, gate = self.proj(x).chunk(2, dim=-1)
return x * F.gelu(gate)
class FeedForward(nn.Module):
def __init__(self, dim, dim_out=None, mult=4, glu=False, dropout=0.):
super().__init__()
inner_dim = int(dim * mult)
dim_out = default(dim_out, dim)
project_in = nn.Sequential(
nn.Linear(dim, inner_dim),
nn.GELU()
) if not glu else GEGLU(dim, inner_dim)
self.net = nn.Sequential(
project_in,
nn.Dropout(dropout),
nn.Linear(inner_dim, dim_out)
)
def forward(self, x):
return self.net(x)
# attention.
class Attention(nn.Module):
def __init__(
self,
dim,
dim_head=DEFAULT_DIM_HEAD,
heads=8,
causal=False,
mask=None,
talking_heads=False,
sparse_topk=None,
use_entmax15=False,
num_mem_kv=0,
dropout=0.,
on_attn=False
):
super().__init__()
if use_entmax15:
raise NotImplementedError("Check out entmax activation instead of softmax activation!")
self.scale = dim_head ** -0.5
self.heads = heads
self.causal = causal
self.mask = mask
inner_dim = dim_head * heads
self.to_q = nn.Linear(dim, inner_dim, bias=False)
self.to_k = nn.Linear(dim, inner_dim, bias=False)
self.to_v = nn.Linear(dim, inner_dim, bias=False)
self.dropout = nn.Dropout(dropout)
# talking heads
self.talking_heads = talking_heads
if talking_heads:
self.pre_softmax_proj = nn.Parameter(torch.randn(heads, heads))
self.post_softmax_proj = nn.Parameter(torch.randn(heads, heads))
# explicit topk sparse attention
self.sparse_topk = sparse_topk
# entmax
#self.attn_fn = entmax15 if use_entmax15 else F.softmax
self.attn_fn = F.softmax
# add memory key / values
self.num_mem_kv = num_mem_kv
if num_mem_kv > 0:
self.mem_k = nn.Parameter(torch.randn(heads, num_mem_kv, dim_head))
self.mem_v = nn.Parameter(torch.randn(heads, num_mem_kv, dim_head))
# attention on attention
self.attn_on_attn = on_attn
self.to_out = nn.Sequential(nn.Linear(inner_dim, dim * 2), nn.GLU()) if on_attn else nn.Linear(inner_dim, dim)
def forward(
self,
x,
context=None,
mask=None,
context_mask=None,
rel_pos=None,
sinusoidal_emb=None,
prev_attn=None,
mem=None
):
b, n, _, h, talking_heads, device = *x.shape, self.heads, self.talking_heads, x.device
kv_input = default(context, x)
q_input = x
k_input = kv_input
v_input = kv_input
if exists(mem):
k_input = torch.cat((mem, k_input), dim=-2)
v_input = torch.cat((mem, v_input), dim=-2)
if exists(sinusoidal_emb):
# in shortformer, the query would start at a position offset depending on the past cached memory
offset = k_input.shape[-2] - q_input.shape[-2]
q_input = q_input + sinusoidal_emb(q_input, offset=offset)
k_input = k_input + sinusoidal_emb(k_input)
q = self.to_q(q_input)
k = self.to_k(k_input)
v = self.to_v(v_input)
q, k, v = map(lambda t: rearrange(t, 'b n (h d) -> b h n d', h=h), (q, k, v))
input_mask = None
if any(map(exists, (mask, context_mask))):
q_mask = default(mask, lambda: torch.ones((b, n), device=device).bool())
k_mask = q_mask if not exists(context) else context_mask
k_mask = default(k_mask, lambda: torch.ones((b, k.shape[-2]), device=device).bool())
q_mask = rearrange(q_mask, 'b i -> b () i ()')
k_mask = rearrange(k_mask, 'b j -> b () () j')
input_mask = q_mask * k_mask
if self.num_mem_kv > 0:
mem_k, mem_v = map(lambda t: repeat(t, 'h n d -> b h n d', b=b), (self.mem_k, self.mem_v))
k = torch.cat((mem_k, k), dim=-2)
v = torch.cat((mem_v, v), dim=-2)
if exists(input_mask):
input_mask = F.pad(input_mask, (self.num_mem_kv, 0), value=True)
dots = einsum('b h i d, b h j d -> b h i j', q, k) * self.scale
mask_value = max_neg_value(dots)
if exists(prev_attn):
dots = dots + prev_attn
pre_softmax_attn = dots
if talking_heads:
dots = einsum('b h i j, h k -> b k i j', dots, self.pre_softmax_proj).contiguous()
if exists(rel_pos):
dots = rel_pos(dots)
if exists(input_mask):
dots.masked_fill_(~input_mask, mask_value)
del input_mask
if self.causal:
i, j = dots.shape[-2:]
r = torch.arange(i, device=device)
mask = rearrange(r, 'i -> () () i ()') < rearrange(r, 'j -> () () () j')
mask = F.pad(mask, (j - i, 0), value=False)
dots.masked_fill_(mask, mask_value)
del mask
if exists(self.sparse_topk) and self.sparse_topk < dots.shape[-1]:
top, _ = dots.topk(self.sparse_topk, dim=-1)
vk = top[..., -1].unsqueeze(-1).expand_as(dots)
mask = dots < vk
dots.masked_fill_(mask, mask_value)
del mask
attn = self.attn_fn(dots, dim=-1)
post_softmax_attn = attn
attn = self.dropout(attn)
if talking_heads:
attn = einsum('b h i j, h k -> b k i j', attn, self.post_softmax_proj).contiguous()
out = einsum('b h i j, b h j d -> b h i d', attn, v)
out = rearrange(out, 'b h n d -> b n (h d)')
intermediates = Intermediates(
pre_softmax_attn=pre_softmax_attn,
post_softmax_attn=post_softmax_attn
)
return self.to_out(out), intermediates
class AttentionLayers(nn.Module):
def __init__(
self,
dim,
depth,
heads=8,
causal=False,
cross_attend=False,
only_cross=False,
use_scalenorm=False,
use_rmsnorm=False,
use_rezero=False,
rel_pos_num_buckets=32,
rel_pos_max_distance=128,
position_infused_attn=False,
custom_layers=None,
sandwich_coef=None,
par_ratio=None,
residual_attn=False,
cross_residual_attn=False,
macaron=False,
pre_norm=True,
gate_residual=False,
**kwargs
):
super().__init__()
ff_kwargs, kwargs = groupby_prefix_and_trim('ff_', kwargs)
attn_kwargs, _ = groupby_prefix_and_trim('attn_', kwargs)
dim_head = attn_kwargs.get('dim_head', DEFAULT_DIM_HEAD)
self.dim = dim
self.depth = depth
self.layers = nn.ModuleList([])
self.has_pos_emb = position_infused_attn
self.pia_pos_emb = FixedPositionalEmbedding(dim) if position_infused_attn else None
self.rotary_pos_emb = always(None)
assert rel_pos_num_buckets <= rel_pos_max_distance, 'number of relative position buckets must be less than the relative position max distance'
self.rel_pos = None
self.pre_norm = pre_norm
self.residual_attn = residual_attn
self.cross_residual_attn = cross_residual_attn
norm_class = ScaleNorm if use_scalenorm else nn.LayerNorm
norm_class = RMSNorm if use_rmsnorm else norm_class
norm_fn = partial(norm_class, dim)
norm_fn = nn.Identity if use_rezero else norm_fn
branch_fn = Rezero if use_rezero else None
if cross_attend and not only_cross:
default_block = ('a', 'c', 'f')
elif cross_attend and only_cross:
default_block = ('c', 'f')
else:
default_block = ('a', 'f')
if macaron:
default_block = ('f',) + default_block
if exists(custom_layers):
layer_types = custom_layers
elif exists(par_ratio):
par_depth = depth * len(default_block)
assert 1 < par_ratio <= par_depth, 'par ratio out of range'
default_block = tuple(filter(not_equals('f'), default_block))
par_attn = par_depth // par_ratio
depth_cut = par_depth * 2 // 3 # 2 / 3 attention layer cutoff suggested by PAR paper
par_width = (depth_cut + depth_cut // par_attn) // par_attn
assert len(default_block) <= par_width, 'default block is too large for par_ratio'
par_block = default_block + ('f',) * (par_width - len(default_block))
par_head = par_block * par_attn
layer_types = par_head + ('f',) * (par_depth - len(par_head))
elif exists(sandwich_coef):
assert sandwich_coef > 0 and sandwich_coef <= depth, 'sandwich coefficient should be less than the depth'
layer_types = ('a',) * sandwich_coef + default_block * (depth - sandwich_coef) + ('f',) * sandwich_coef
else:
layer_types = default_block * depth
self.layer_types = layer_types
self.num_attn_layers = len(list(filter(equals('a'), layer_types)))
for layer_type in self.layer_types:
if layer_type == 'a':
layer = Attention(dim, heads=heads, causal=causal, **attn_kwargs)
elif layer_type == 'c':
layer = Attention(dim, heads=heads, **attn_kwargs)
elif layer_type == 'f':
layer = FeedForward(dim, **ff_kwargs)
layer = layer if not macaron else Scale(0.5, layer)
else:
raise Exception(f'invalid layer type {layer_type}')
if isinstance(layer, Attention) and exists(branch_fn):
layer = branch_fn(layer)
if gate_residual:
residual_fn = GRUGating(dim)
else:
residual_fn = Residual()
self.layers.append(nn.ModuleList([
norm_fn(),
layer,
residual_fn
]))
def forward(
self,
x,
context=None,
mask=None,
context_mask=None,
mems=None,
return_hiddens=False
):
hiddens = []
intermediates = []
prev_attn = None
prev_cross_attn = None
mems = mems.copy() if exists(mems) else [None] * self.num_attn_layers
for ind, (layer_type, (norm, block, residual_fn)) in enumerate(zip(self.layer_types, self.layers)):
is_last = ind == (len(self.layers) - 1)
if layer_type == 'a':
hiddens.append(x)
layer_mem = mems.pop(0)
residual = x
if self.pre_norm:
x = norm(x)
if layer_type == 'a':
out, inter = block(x, mask=mask, sinusoidal_emb=self.pia_pos_emb, rel_pos=self.rel_pos,
prev_attn=prev_attn, mem=layer_mem)
elif layer_type == 'c':
out, inter = block(x, context=context, mask=mask, context_mask=context_mask, prev_attn=prev_cross_attn)
elif layer_type == 'f':
out = block(x)
x = residual_fn(out, residual)
if layer_type in ('a', 'c'):
intermediates.append(inter)
if layer_type == 'a' and self.residual_attn:
prev_attn = inter.pre_softmax_attn
elif layer_type == 'c' and self.cross_residual_attn:
prev_cross_attn = inter.pre_softmax_attn
if not self.pre_norm and not is_last:
x = norm(x)
if return_hiddens:
intermediates = LayerIntermediates(
hiddens=hiddens,
attn_intermediates=intermediates
)
return x, intermediates
return x
class Encoder(AttentionLayers):
def __init__(self, **kwargs):
assert 'causal' not in kwargs, 'cannot set causality on encoder'
super().__init__(causal=False, **kwargs)
class TransformerWrapper(nn.Module):
def __init__(
self,
*,
num_tokens,
max_seq_len,
attn_layers,
emb_dim=None,
max_mem_len=0.,
emb_dropout=0.,
num_memory_tokens=None,
tie_embedding=False,
use_pos_emb=True
):
super().__init__()
assert isinstance(attn_layers, AttentionLayers), 'attention layers must be one of Encoder or Decoder'
dim = attn_layers.dim
emb_dim = default(emb_dim, dim)
self.max_seq_len = max_seq_len
self.max_mem_len = max_mem_len
self.num_tokens = num_tokens
self.token_emb = nn.Embedding(num_tokens, emb_dim)
self.pos_emb = AbsolutePositionalEmbedding(emb_dim, max_seq_len) if (
use_pos_emb and not attn_layers.has_pos_emb) else always(0)
self.emb_dropout = nn.Dropout(emb_dropout)
self.project_emb = nn.Linear(emb_dim, dim) if emb_dim != dim else nn.Identity()
self.attn_layers = attn_layers
self.norm = nn.LayerNorm(dim)
self.init_()
self.to_logits = nn.Linear(dim, num_tokens) if not tie_embedding else lambda t: t @ self.token_emb.weight.t()
# memory tokens (like [cls]) from Memory Transformers paper
num_memory_tokens = default(num_memory_tokens, 0)
self.num_memory_tokens = num_memory_tokens
if num_memory_tokens > 0:
self.memory_tokens = nn.Parameter(torch.randn(num_memory_tokens, dim))
# let funnel encoder know number of memory tokens, if specified
if hasattr(attn_layers, 'num_memory_tokens'):
attn_layers.num_memory_tokens = num_memory_tokens
def init_(self):
nn.init.normal_(self.token_emb.weight, std=0.02)
def forward(
self,
x,
return_embeddings=False,
mask=None,
return_mems=False,
return_attn=False,
mems=None,
**kwargs
):
b, n, device, num_mem = *x.shape, x.device, self.num_memory_tokens
x = self.token_emb(x)
x += self.pos_emb(x)
x = self.emb_dropout(x)
x = self.project_emb(x)
if num_mem > 0:
mem = repeat(self.memory_tokens, 'n d -> b n d', b=b)
x = torch.cat((mem, x), dim=1)
# auto-handle masking after appending memory tokens
if exists(mask):
mask = F.pad(mask, (num_mem, 0), value=True)
x, intermediates = self.attn_layers(x, mask=mask, mems=mems, return_hiddens=True, **kwargs)
x = self.norm(x)
mem, x = x[:, :num_mem], x[:, num_mem:]
out = self.to_logits(x) if not return_embeddings else x
if return_mems:
hiddens = intermediates.hiddens
new_mems = list(map(lambda pair: torch.cat(pair, dim=-2), zip(mems, hiddens))) if exists(mems) else hiddens
new_mems = list(map(lambda t: t[..., -self.max_mem_len:, :].detach(), new_mems))
return out, new_mems
if return_attn:
attn_maps = list(map(lambda t: t.post_softmax_attn, intermediates.attn_intermediates))
return out, attn_maps
return out
+353
View File
@@ -0,0 +1,353 @@
import argparse
import datetime
import glob
import json
import math
import os
import sys
import time
from collections import OrderedDict
import cv2
import numpy as np
import torch
import torchvision
## note: decord should be imported after torch
from omegaconf import OmegaConf
from pytorch_lightning import seed_everything
from tqdm import tqdm
sys.path.insert(1, os.path.join(sys.path[0], '..', '..'))
from lvdm.models.samplers.ddim import DDIMSampler
from main.evaluation.motionctrl_prompts_camerapose_trajs import (
both_prompt_camerapose_traj, cmcm_prompt_camerapose, omom_prompt_traj)
from utils.utils import instantiate_from_config
DEFAULT_NEGATIVE_PROMPT = 'blur, haze, deformed iris, deformed pupils, semi-realistic, cgi, 3d, render, '\
'sketch, cartoon, drawing, anime, mutated hands and fingers, deformed, distorted, '\
'disfigured, poorly drawn, bad anatomy, wrong anatomy, extra limb, missing limb, '\
'floating limbs, disconnected limbs, mutation, mutated, ugly, disgusting, amputation'
post_prompt = 'Ultra-detail, masterpiece, best quality, cinematic lighting, 8k uhd, dslr, soft lighting, film grain, Fujifilm XT3'
def load_model_checkpoint(model, ckpt, adapter_ckpt=None):
if adapter_ckpt:
## main model
state_dict = torch.load(ckpt, map_location="cpu")
if "state_dict" in list(state_dict.keys()):
state_dict = state_dict["state_dict"]
result = model.load_state_dict(state_dict, strict=False)
else:
# deepspeed
new_pl_sd = OrderedDict()
for key in state_dict['module'].keys():
new_pl_sd[key[16:]]=state_dict['module'][key]
result = model.load_state_dict(new_pl_sd, strict=False)
print(result)
print('>>> model checkpoint loaded.')
## adapter
state_dict = torch.load(adapter_ckpt, map_location="cpu")
if "state_dict" in list(state_dict.keys()):
state_dict = state_dict["state_dict"]
model.adapter.load_state_dict(state_dict, strict=True)
print('>>> adapter checkpoint loaded.')
else:
state_dict = torch.load(ckpt, map_location="cpu")
if "state_dict" in list(state_dict.keys()):
state_dict = state_dict["state_dict"]
model.load_state_dict(state_dict, strict=False)
else:
# deepspeed
new_pl_sd = OrderedDict()
for key in state_dict['module'].keys():
new_pl_sd[key[16:]]=state_dict['module'][key]
model.load_state_dict(new_pl_sd)
print('>>> model checkpoint loaded.')
return model
def load_trajs(cond_dir, trajs):
traj_files = [f'{cond_dir}/trajectories/{traj}.npy' for traj in trajs]
data_list = []
traj_name = []
for idx in range(len(traj_files)):
traj_name.append(traj_files[idx].split('/')[-1].split('.')[0])
data_list.append(torch.tensor(np.load(traj_files[idx])).permute(3, 0, 1, 2).float()) # [t,h,w,c] -> [c,t,h,w]
return data_list, traj_name
def load_camera_pose(cond_dir, camera_poses):
pose_file = [f'{cond_dir}/camera_poses/{pose}.json' for pose in camera_poses]
pose_sample_num = len(pose_file)
data_list = []
pose_name = []
for idx in range(pose_sample_num):
cur_pose_name = camera_poses[idx].replace('test_camera_', '')
pose_name.append(cur_pose_name)
with open(pose_file[idx], 'r') as f:
pose = json.load(f)
pose = np.array(pose) # [t, 12]
pose = torch.tensor(pose).float() # [t, 12]
data_list.append(pose)
return data_list, pose_name
def save_results(samples, filename, savedir, fps=10):
## save prompt
## save video
videos = [samples]
savedirs = [savedir]
for idx, video in enumerate(videos):
if video is None:
continue
# b,c,t,h,w
video = video.detach().cpu()
video = torch.clamp(video.float(), -1., 1.)
n = video.shape[0]
video = video.permute(2, 0, 1, 3, 4) # t,n,c,h,w
frame_grids = [torchvision.utils.make_grid(framesheet, nrow=int(n)) for framesheet in video] #[3, 1*h, n*w]
grid = torch.stack(frame_grids, dim=0) # stack in temporal dim [t, 3, n*h, w]
grid = (grid + 1.0) / 2.0
grid = (grid * 255).to(torch.uint8).permute(0, 2, 3, 1)
path = os.path.join(savedirs[idx], "%s.mp4"%filename)
torchvision.io.write_video(path, grid, fps=fps, video_codec='h264', options={'crf': '10'})
def motionctrl_sample(
model,
prompts,
noise_shape,
camera_poses=None,
trajs=None,
n_samples=1,
unconditional_guidance_scale=1.0,
unconditional_guidance_scale_temporal=None,
ddim_steps=50,
ddim_eta=1.,
**kwargs):
ddim_sampler = DDIMSampler(model)
batch_size = noise_shape[0]
## get condition embeddings (support single prompt only)
if isinstance(prompts, str):
prompts = [prompts]
for i in range(len(prompts)):
prompts[i] = f'{prompts[i]}, {post_prompt}'
cond = model.get_learned_conditioning(prompts)
if camera_poses is not None:
RT = camera_poses[..., None]
else:
RT = None
if trajs is not None:
traj_features = model.get_traj_features(trajs)
else:
traj_features = None
if unconditional_guidance_scale != 1.0:
# prompts = batch_size * [""]
prompts = batch_size * [DEFAULT_NEGATIVE_PROMPT]
uc = model.get_learned_conditioning(prompts)
if traj_features is not None:
un_motion = model.get_traj_features(torch.zeros_like(trajs))
else:
un_motion = None
uc = {"features_adapter": un_motion, "uc": uc}
else:
uc = None
batch_variants = []
for _ in range(n_samples):
if ddim_sampler is not None:
samples, _ = ddim_sampler.sample(S=ddim_steps,
conditioning=cond,
batch_size=noise_shape[0],
shape=noise_shape[1:],
verbose=False,
unconditional_guidance_scale=unconditional_guidance_scale,
unconditional_conditioning=uc,
eta=ddim_eta,
temporal_length=noise_shape[2],
conditional_guidance_scale_temporal=unconditional_guidance_scale_temporal,
features_adapter=traj_features,
pose_emb=RT,
**kwargs
)
## reconstruct from latent to pixel space
batch_images = model.decode_first_stage(samples)
batch_variants.append(batch_images)
## variants, batch, c, t, h, w
batch_variants = torch.stack(batch_variants)
return batch_variants.permute(1, 0, 2, 3, 4, 5)
def run_inference(args, gpu_num, gpu_no):
## model config
config = OmegaConf.load(args.base)
model_config = config.pop("model", OmegaConf.create())
model = instantiate_from_config(model_config)
model = model.cuda(gpu_no)
assert os.path.exists(args.ckpt_path), f"Error: checkpoint {args.ckpt_path} Not Found!"
print(f"Loading checkpoint from {args.ckpt_path}")
model = load_model_checkpoint(model, args.ckpt_path, args.adapter_ckpt)
model.eval()
## run over data
assert (args.height % 16 == 0) and (args.width % 16 == 0), "Error: image size [h,w] should be multiples of 16!"
## latent noise shape
h, w = args.height // 8, args.width // 8
channels = model.channels
frames = model.temporal_length
noise_shape = [args.bs, channels, frames, h, w]
savedir = os.path.join(args.savedir, "samples")
os.makedirs(savedir, exist_ok=True)
if args.condtype == 'camera_motion':
prompt_list = cmcm_prompt_camerapose['prompts']
camera_pose_list, pose_name = load_camera_pose(args.cond_dir, cmcm_prompt_camerapose['camera_poses'])
traj_list = None
save_name_list = []
for i in range(len(pose_name)):
save_name_list.append(f"{pose_name[i]}__{prompt_list[i].replace(' ', '_').replace(',', '')}")
elif args.condtype == 'object_motion':
prompt_list = omom_prompt_traj['prompts']
traj_list, traj_name = load_trajs(args.cond_dir, omom_prompt_traj['trajs'])
camera_pose_list = None
save_name_list = []
for i in range(len(traj_name)):
save_name_list.append(f"{traj_name[i]}__{prompt_list[i].replace(' ', '_').replace(',', '')}")
elif args.condtype == 'both':
prompt_list = both_prompt_camerapose_traj['prompts']
camera_pose_list, pose_name = load_camera_pose(args.cond_dir, both_prompt_camerapose_traj['camera_poses'])
traj_list, traj_name = load_trajs(args.cond_dir, both_prompt_camerapose_traj['trajs'])
save_name_list = []
for i in range(len(pose_name)):
save_name_list.append(f"{pose_name[i]}__{traj_name[i]}__{prompt_list[i].replace(' ', '_').replace(',', '')}")
num_samples = len(prompt_list)
samples_split = num_samples // gpu_num
print('Prompts testing [rank:%d] %d/%d samples loaded.'%(gpu_no, samples_split, num_samples))
#indices = random.choices(list(range(0, num_samples)), k=samples_per_device)
indices = list(range(samples_split*gpu_no, samples_split*(gpu_no+1)))
prompt_list_rank = [prompt_list[i] for i in indices]
camera_pose_list_rank = None if camera_pose_list is None else [camera_pose_list[i] for i in indices]
traj_list_rank = None if traj_list is None else [traj_list[i] for i in indices]
save_name_list_rank = [save_name_list[i] for i in indices]
start = time.time()
for idx, indice in tqdm(enumerate(range(0, len(prompt_list_rank), args.bs)), desc='Sample Batch'):
prompts = prompt_list_rank[indice:indice+args.bs]
camera_poses = None if camera_pose_list_rank is None else camera_pose_list_rank[indice:indice+args.bs]
trajs = None if traj_list_rank is None else traj_list_rank[indice:indice+args.bs]
save_name = save_name_list_rank[indice:indice+args.bs]
print(f'Processing {save_name}')
if camera_poses is not None:
camera_poses = torch.stack(camera_poses, dim=0).to("cuda")
if trajs is not None:
trajs = torch.stack(trajs, dim=0).to("cuda")
batch_samples = motionctrl_sample(
model,
prompts,
noise_shape,
camera_poses=camera_poses,
trajs=trajs,
n_samples=args.n_samples,
unconditional_guidance_scale=args.unconditional_guidance_scale,
unconditional_guidance_scale_temporal=args.unconditional_guidance_scale_temporal,
ddim_steps=args.ddim_steps,
ddim_eta=args.ddim_eta,
cond_T = args.cond_T,
)
## save each example individually
for nn, samples in enumerate(batch_samples):
## samples : [n_samples,c,t,h,w]
prompt = prompts[nn]
name = save_name[nn]
if len(name) > 90:
name = name[:90]
filename = f'{name}_{idx*args.bs+nn:04d}_randk{gpu_no}'
save_results(samples, filename, savedir, fps=10)
if args.save_imgs:
parts = save_name[nn].split('__')
if len(parts) == 2:
cond_name = parts[0]
prname = prompts[nn].replace(' ', '_').replace(',', '')
cur_outdir = os.path.join(savedir, cond_name, prname)
elif len(parts) == 3:
poname, trajname, _ = save_name[nn].split('__')
prname = prompts[nn].replace(' ', '_').replace(',', '')
cur_outdir = os.path.join(savedir, poname, trajname, prname)
else:
raise NotImplementedError
os.makedirs(cur_outdir, exist_ok=True)
save_images(samples, cur_outdir)
if nn % 100 == 0:
print(f'Finish {nn}/{len(batch_samples)}')
print(f"Saved in {args.savedir}. Time used: {(time.time() - start):.2f} seconds")
def save_images(samples, savedir):
## samples : [n_samples,c,t,h,w]
n_samples, c, t, h, w = samples.shape
samples = torch.clamp(samples, -1.0, 1.0)
samples = (samples + 1.0) / 2.0
samples = (samples * 255).detach().cpu().numpy().astype(np.uint8)
for i in range(n_samples):
cur_outdir = os.path.join(savedir, f'{i}/images')
os.makedirs(cur_outdir, exist_ok=True)
for j in range(t):
img = samples[i,:,j,:,:]
img = np.transpose(img, (1,2,0))
img = img[:,:,::-1] # BGR to RGB
path = os.path.join(cur_outdir, f'{j:04d}.png')
cv2.imwrite(path, img)
def get_parser():
parser = argparse.ArgumentParser()
parser.add_argument("--savedir", type=str, default=None, help="results saving path")
parser.add_argument("--ckpt_path", type=str, default=None, help="checkpoint path")
parser.add_argument("--adapter_ckpt", type=str, default=None, help="adapter checkpoint path")
parser.add_argument("--base", type=str, help="config (yaml) path")
parser.add_argument("--condtype", default='frame', type=str, help="conditon type: {frame, depth, adapter}")
parser.add_argument("--prompt_dir", type=str, default=None, help="a data dir containing videos and prompts")
parser.add_argument("--n_samples", type=int, default=1, help="num of samples per prompt",)
parser.add_argument("--ddim_steps", type=int, default=50, help="steps of ddim if positive, otherwise use DDPM",)
parser.add_argument("--ddim_eta", type=float, default=1.0, help="eta for ddim sampling (0.0 yields deterministic sampling)",)
parser.add_argument("--bs", type=int, default=1, help="batch size for inference")
parser.add_argument("--height", type=int, default=512, help="image height, in pixel space")
parser.add_argument("--width", type=int, default=512, help="image width, in pixel space")
parser.add_argument("--unconditional_guidance_scale", type=float, default=1.0, help="prompt classifier-free guidance")
parser.add_argument("--unconditional_guidance_scale_temporal", type=float, default=None, help="temporal consistency guidance")
parser.add_argument("--seed", type=int, default=20230211, help="seed for seed_everything")
parser.add_argument("--cond_T", default=800, type=int, help="Steps smaller than cond_T will not contain condition")
parser.add_argument("--save_imgs", action='store_true', help="save condition")
parser.add_argument("--cond_dir", type=str, default=None, help="condition dir")
return parser
if __name__ == '__main__':
now = datetime.datetime.now().strftime("%Y-%m-%d-%H-%M-%S")
print("@CoLVDM cond-Inference: %s"%now)
parser = get_parser()
args, unkown = parser.parse_known_args()
# args = parser.parse_args()
seed_everything(args.seed)
rank, gpu_num = 0, 1
run_inference(args, gpu_num, rank)
@@ -0,0 +1,114 @@
##### CMCM #####
complex_camera_poses = [
"test_camera_d971457c81bca597",
"test_camera_d971457c81bca597",
"test_camera_d971457c81bca597",
'test_camera_Round-ZoomIn',
'test_camera_Round-ZoomIn',
'test_camera_Round-ZoomIn'
]
complex_camera_pose_prompt = [
"a temple on a mountain, bird's view",
"Effiel Tower in Paris, bird's view",
"a castle in a forest, bird's view",
"a temple on a mountain, bird's view",
"Effiel Tower in Paris, bird's view",
"a castle in a forest, bird's view",
]
basic_camera_poses = [
'test_camera_L',
'test_camera_D',
'test_camera_I',
'test_camera_O',
'test_camera_R',
'test_camera_U',
'test_camera_SPIN-CW-60',
'test_camera_SPIN-ACW-60',
]
basic_camera_pose_prompt = [
'coastline, rocks, storm weather, wind, waves, lightning',
'coastline, rocks, storm weather, wind, waves, lightning',
'coastline, rocks, storm weather, wind, waves, lightning',
'coastline, rocks, storm weather, wind, waves, lightning',
'coastline, rocks, storm weather, wind, waves, lightning',
'coastline, rocks, storm weather, wind, waves, lightning',
'coastline, rocks, storm weather, wind, waves, lightning',
'coastline, rocks, storm weather, wind, waves, lightning',
]
diff_speeds_camera_poses = [
'test_camera_I_0.2x',
'test_camera_I_0.4x',
'test_camera_I_1.0x',
'test_camera_I_2.0x',
'test_camera_O_0.2x',
'test_camera_O_0.4x',
'test_camera_O_1.0x',
'test_camera_O_2.0x',
]
diff_speeds_camera_pose_prompt = [
'A sunrise landscape features mountains and lakes',
'A sunrise landscape features mountains and lakes',
'A sunrise landscape features mountains and lakes',
'A sunrise landscape features mountains and lakes',
'A sunrise landscape features mountains and lakes',
'A sunrise landscape features mountains and lakes',
'A sunrise landscape features mountains and lakes',
'A sunrise landscape features mountains and lakes',
]
cmcm_prompt_camerapose = {
'prompts': complex_camera_pose_prompt + basic_camera_pose_prompt + diff_speeds_camera_pose_prompt,
'camera_poses': complex_camera_poses + basic_camera_poses + diff_speeds_camera_poses
}
assert len(cmcm_prompt_camerapose['prompts']) == len(cmcm_prompt_camerapose['camera_poses']), \
"The number of prompts and camera poses should be the same."
### OMCM ###
trajs = [
'shake_1', 'shake_1', 'shake_1',
'curve_2', 'curve_2', 'curve_2',
]
traj_prompt = [
'a sunflower swaying in the wind',
'a rose swaying in the wind',
'a wind chime swaying in the wind',
'a man surfing',
'a man skateboarding',
'a girl skiing'
]
omom_prompt_traj = {
'prompts': traj_prompt,
'trajs': trajs
}
assert len(omom_prompt_traj['prompts']) == len(omom_prompt_traj['trajs']), \
"The number of prompts and trajs should be the same."
both_camerapose = [
'test_camera_O'
]
both_traj = [
'shaking_10'
]
both_prompt = [
'a rose swaying in the wind'
]
both_prompt_camerapose_traj = {
'prompts': both_prompt,
'camera_poses': both_camerapose,
'trajs': both_traj
}
assert len(both_prompt_camerapose_traj['prompts']) == len(both_prompt_camerapose_traj['camera_poses']) == len(both_prompt_camerapose_traj['trajs']), \
"The number of prompts, camera poses and trajs should be the same."
+153
View File
@@ -0,0 +1,153 @@
import logging
import torch
from einops import rearrange, repeat
from lvdm.models.utils_diffusion import timestep_embedding
try:
import xformers
import xformers.ops
XFORMERS_IS_AVAILBLE = True
except:
XFORMERS_IS_AVAILBLE = False
mainlogger = logging.getLogger('mainlogger')
def TemporalTransformer_forward(self, x, context=None, is_imgbatch=False):
b, c, t, h, w = x.shape
x_in = x
x = self.norm(x)
x = rearrange(x, 'b c t h w -> (b h w) c t').contiguous()
if not self.use_linear:
x = self.proj_in(x)
x = rearrange(x, 'bhw c t -> bhw t c').contiguous()
if self.use_linear:
x = self.proj_in(x)
temp_mask = None
if self.causal_attention:
temp_mask = torch.tril(torch.ones([1, t, t]))
if is_imgbatch:
temp_mask = torch.eye(t).unsqueeze(0)
if temp_mask is not None:
mask = temp_mask.to(x.device)
mask = repeat(mask, 'l i j -> (l bhw) i j', bhw=b*h*w)
else:
mask = None
if self.only_self_att:
## note: if no context is given, cross-attention defaults to self-attention
for i, block in enumerate(self.transformer_blocks):
x = block(x, context=context, mask=mask)
x = rearrange(x, '(b hw) t c -> b hw t c', b=b).contiguous()
else:
x = rearrange(x, '(b hw) t c -> b hw t c', b=b).contiguous()
context = rearrange(context, '(b t) l con -> b t l con', t=t).contiguous()
for i, block in enumerate(self.transformer_blocks):
# calculate each batch one by one (since number in shape could not greater then 65,535 for some package)
for j in range(b):
unit_context = context[j][0:1]
context_j = repeat(unit_context, 't l con -> (t r) l con', r=(h * w)).contiguous()
## note: causal mask will not applied in cross-attention case
x[j] = block(x[j], context=context_j)
if self.use_linear:
x = self.proj_out(x)
x = rearrange(x, 'b (h w) t c -> b c t h w', h=h, w=w).contiguous()
if not self.use_linear:
x = rearrange(x, 'b hw t c -> (b hw) c t').contiguous()
x = self.proj_out(x)
x = rearrange(x, '(b h w) c t -> b c t h w', b=b, h=h, w=w).contiguous()
if self.use_image_dataset:
x = 0.0 * x + x_in
else:
x = x + x_in
return x
def selfattn_forward_unet(self, x, timesteps, context=None, y=None, features_adapter=None, is_imgbatch=False, T=None, **kwargs):
b,_,t,_,_ = x.shape
t_emb = timestep_embedding(timesteps, self.model_channels, repeat_only=False)
emb = self.time_embed(t_emb)
if self.micro_condition and y is not None:
micro_emb = timestep_embedding(y, self.model_channels, repeat_only=False)
emb = emb + self.micro_embed(micro_emb)
# pose_emb = pose_emb.reshape(-1, pose_emb.shape[-1])
## repeat t times for context [(b t) 77 768] & time embedding
if not is_imgbatch:
context = context.repeat_interleave(repeats=t, dim=0)
if 'pose_emb' in kwargs:
pose_emb = kwargs.pop('pose_emb')
context = { 'context': context, 'pose_emb': pose_emb }
emb = emb.repeat_interleave(repeats=t, dim=0)
## always in shape (b t) c h w, except for temporal layer
x = rearrange(x, 'b c t h w -> (b t) c h w')
if features_adapter is not None:
features_adapter = [rearrange(feature, 'b c t h w -> (b t) c h w') for feature in features_adapter]
h = x.type(self.dtype)
adapter_idx = 0
hs = []
for id, module in enumerate(self.input_blocks):
h = module(h, emb, context=context, batch_size=b,is_imgbatch=is_imgbatch)
if id ==0 and self.addition_attention:
h = self.init_attn(h, emb, context=context, batch_size=b,is_imgbatch=is_imgbatch)
## plug-in adapter features
if ((id+1)%3 == 0) and features_adapter is not None:
# if adapter_idx == 0 or adapter_idx == 1 or adapter_idx == 2:
h = h + features_adapter[adapter_idx]
adapter_idx += 1
hs.append(h)
if features_adapter is not None:
assert len(features_adapter)==adapter_idx, 'Wrong features_adapter'
h = self.middle_block(h, emb, context=context, batch_size=b, is_imgbatch=is_imgbatch)
for module in self.output_blocks:
h = torch.cat([h, hs.pop()], dim=1)
h = module(h, emb, context=context, batch_size=b, is_imgbatch=is_imgbatch)
h = h.type(x.dtype)
y = self.out(h)
# reshape back to (b c t h w)
y = rearrange(y, '(b t) c h w -> b c t h w', b=b)
return y
def spatial_forward_BasicTransformerBlock(self, x, context=None, mask=None):
if isinstance(context, dict):
context = context['context']
x = self.attn1(self.norm1(x), context=context if self.disable_self_attn else None, mask=mask) + x
x = self.attn2(self.norm2(x), context=context, mask=mask) + x
x = self.ff(self.norm3(x)) + x
return x
def temporal_selfattn_forward_BasicTransformerBlock(self, x, context=None, mask=None):
if isinstance(context, dict) and 'pose_emb' in context:
pose_emb = context['pose_emb'] # {channel_num: [B, video_length, pose_dim, pose_embedding_dim]}
context = None
else:
pose_emb = None
context = None
x = self.attn1(self.norm1(x), context=context if self.disable_self_attn else None, mask=mask) + x
# Add camera pose
if pose_emb is not None:
B, t, _, _ = pose_emb.shape # [B, video_length, pose_dim, pose_embedding_dim]
hw = x.shape[0] // B
pose_emb = pose_emb.reshape(B, t, -1)
pose_emb = pose_emb.repeat_interleave(repeats=hw, dim=0)
x = self.cc_projection(torch.cat([x, pose_emb], dim=-1))
x = self.attn2(self.norm2(x), context=context, mask=mask) + x
x = self.ff(self.norm3(x)) + x
return x
+67
View File
@@ -0,0 +1,67 @@
import torch.nn as nn
from einops import rearrange
from lvdm.models.ddpm3d import LatentDiffusion
from motionctrl.lvdm_modified_modules import (
TemporalTransformer_forward, selfattn_forward_unet,
spatial_forward_BasicTransformerBlock,
temporal_selfattn_forward_BasicTransformerBlock)
from utils.utils import instantiate_from_config
class MotionCtrl(LatentDiffusion):
def __init__(self,
omcm_config=None,
pose_dim=12,
context_dim=1024,
*args,
**kwargs):
super(MotionCtrl, self).__init__(*args, **kwargs)
# object motion control module
if omcm_config is not None:
self.omcm = instantiate_from_config(omcm_config)
else:
self.omcm = None
# camera motion control module
bound_method = selfattn_forward_unet.__get__(
self.model.diffusion_model,
self.model.diffusion_model.__class__)
setattr(self.model.diffusion_model, 'forward', bound_method)
for _name, _module in self.model.diffusion_model.named_modules():
if _module.__class__.__name__ == 'TemporalTransformer':
bound_method = TemporalTransformer_forward.__get__(
_module, _module.__class__)
setattr(_module, 'forward', bound_method)
if _module.__class__.__name__ == 'BasicTransformerBlock':
# SpatialTransformer only
if _module.attn2.to_k.in_features != context_dim: # TemporalTransformer without crossattn
bound_method = temporal_selfattn_forward_BasicTransformerBlock.__get__(
_module, _module.__class__)
setattr(_module, '_forward', bound_method)
cc_projection = nn.Linear(_module.attn2.to_k.in_features + pose_dim, _module.attn2.to_k.in_features)
nn.init.eye_(list(cc_projection.parameters())[0][:_module.attn2.to_k.in_features, :_module.attn2.to_k.in_features])
nn.init.zeros_(list(cc_projection.parameters())[1])
cc_projection.requires_grad_(True)
_module.add_module('cc_projection', cc_projection)
else:
bound_method = spatial_forward_BasicTransformerBlock.__get__(
_module, _module.__class__)
setattr(_module, '_forward', bound_method)
def get_traj_features(self, extra_cond):
b, c, t, h, w = extra_cond.shape
## process in 2D manner
extra_cond = rearrange(extra_cond, 'b c t h w -> (b t) c h w')
traj_features = self.omcm(extra_cond)
traj_features = [rearrange(feature, '(b t) c h w -> b c t h w', b=b, t=t) for feature in traj_features]
return traj_features
+25
View File
@@ -0,0 +1,25 @@
decord==0.6.0
einops==0.3.0
numpy==1.24.2
omegaconf==2.1.1
pandas==2.0.0
Pillow==9.5.0
pytorch_lightning==1.8.3
PyYAML==6.0
setuptools==65.6.3
torch==2.0.0
torchvision
tqdm==4.65.0
transformers==4.25.1
moviepy
av
xformers
timm
scikit-learn
open_clip_torch==2.12.0
kornia
gradio==3.37.0 #3.35.2
plotly
imageio==2.14.1
imageio-ffmpeg==0.4.7
opencv-python==4.8.0.74
+80
View File
@@ -0,0 +1,80 @@
import importlib
import numpy as np
from inspect import isfunction
from PIL import Image, ImageDraw, ImageFont
import cv2
import torch
import torch.distributed as dist
def count_params(model, verbose=False):
total_params = sum(p.numel() for p in model.parameters())
if verbose:
print(f"{model.__class__.__name__} has {total_params*1.e-6:.2f} M params.")
return total_params
def check_istarget(name, para_list):
"""
name: full name of source para
para_list: partial name of target para
"""
istarget=False
for para in para_list:
if para in name:
return True
return istarget
def instantiate_from_config(config):
if not "target" in config:
if config == '__is_first_stage__':
return None
elif config == "__is_unconditional__":
return None
raise KeyError("Expected key `target` to instantiate.")
return get_obj_from_str(config["target"])(**config.get("params", dict()))
def get_obj_from_str(string, reload=False):
module, cls = string.rsplit(".", 1)
if reload:
module_imp = importlib.import_module(module)
importlib.reload(module_imp)
return getattr(importlib.import_module(module, package=None), cls)
def load_npz_from_dir(data_dir):
data = [np.load(os.path.join(data_dir, data_name))['arr_0'] for data_name in os.listdir(data_dir)]
data = np.concatenate(data, axis=0)
return data
def load_npz_from_paths(data_paths):
data = [np.load(data_path)['arr_0'] for data_path in data_paths]
data = np.concatenate(data, axis=0)
return data
def resize_numpy_image(image, max_resolution=512 * 512, resize_short_edge=None):
h, w = image.shape[:2]
if resize_short_edge is not None:
k = resize_short_edge / min(h, w)
else:
k = max_resolution / (h * w)
k = k**0.5
h = int(np.round(h * k / 64)) * 64
w = int(np.round(w * k / 64)) * 64
image = cv2.resize(image, (w, h), interpolation=cv2.INTER_LANCZOS4)
return image
def setup_dist(args):
if dist.is_initialized():
return
torch.cuda.set_device(args.local_rank)
torch.distributed.init_process_group(
'nccl',
init_method='env://'
)