add vision encoder
This commit is contained in:
@@ -0,0 +1,4 @@
|
||||
.venv
|
||||
model
|
||||
Scratchpad.ipynb
|
||||
__pycache__
|
||||
@@ -1,2 +1,16 @@
|
||||
# moondream
|
||||
tiny vision language model
|
||||
|
||||
a tiny vision language model
|
||||
|
||||
## project goals
|
||||
|
||||
Build a high-quality, low-hallucination vision language model small enough to
|
||||
run on an edge device without a GPU.
|
||||
|
||||
## moondream0
|
||||
|
||||
Initial prototype built using SigLIP, Phi-1.5, and the LLaVa training dataset.
|
||||
The model is for research purposes only, and is subject to the Phi and LLaVa
|
||||
license restrictions.
|
||||
|
||||
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 57 KiB |
@@ -0,0 +1,28 @@
|
||||
import torch
|
||||
from PIL import Image
|
||||
from torchvision.transforms.v2 import (
|
||||
Compose,
|
||||
Resize,
|
||||
InterpolationMode,
|
||||
ToImage,
|
||||
ToDtype,
|
||||
Normalize,
|
||||
)
|
||||
|
||||
|
||||
class VisionEncoder:
|
||||
def __init__(self) -> None:
|
||||
self.model = torch.jit.load("model/vision.pt").to(dtype=torch.float32)
|
||||
self.preprocess = Compose(
|
||||
[
|
||||
Resize(size=(384, 384), interpolation=InterpolationMode.BICUBIC),
|
||||
ToImage(),
|
||||
ToDtype(torch.float32, scale=True),
|
||||
Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5]),
|
||||
]
|
||||
)
|
||||
|
||||
def __call__(self, image: Image) -> torch.Tensor:
|
||||
with torch.no_grad():
|
||||
image_vec = self.preprocess(image.convert("RGB")).unsqueeze(0)
|
||||
return self.model(image_vec)
|
||||
@@ -0,0 +1,4 @@
|
||||
torch
|
||||
pillow
|
||||
torchvision
|
||||
transformers
|
||||
Reference in New Issue
Block a user