add vision encoder

This commit is contained in:
vik
2023-12-29 00:21:39 -08:00
parent be7c3e9b93
commit 8842aa770b
5 changed files with 51 additions and 1 deletions
+4
View File
@@ -0,0 +1,4 @@
.venv
model
Scratchpad.ipynb
__pycache__
+15 -1
View File
@@ -1,2 +1,16 @@
# moondream
tiny vision language model
a tiny vision language model
## project goals
Build a high-quality, low-hallucination vision language model small enough to
run on an edge device without a GPU.
## moondream0
Initial prototype built using SigLIP, Phi-1.5, and the LLaVa training dataset.
The model is for research purposes only, and is subject to the Phi and LLaVa
license restrictions.
BIN
View File
Binary file not shown.

After

Width:  |  Height:  |  Size: 57 KiB

+28
View File
@@ -0,0 +1,28 @@
import torch
from PIL import Image
from torchvision.transforms.v2 import (
Compose,
Resize,
InterpolationMode,
ToImage,
ToDtype,
Normalize,
)
class VisionEncoder:
def __init__(self) -> None:
self.model = torch.jit.load("model/vision.pt").to(dtype=torch.float32)
self.preprocess = Compose(
[
Resize(size=(384, 384), interpolation=InterpolationMode.BICUBIC),
ToImage(),
ToDtype(torch.float32, scale=True),
Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5]),
]
)
def __call__(self, image: Image) -> torch.Tensor:
with torch.no_grad():
image_vec = self.preprocess(image.convert("RGB")).unsqueeze(0)
return self.model(image_vec)
+4
View File
@@ -0,0 +1,4 @@
torch
pillow
torchvision
transformers