diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..ee94f05 --- /dev/null +++ b/.gitignore @@ -0,0 +1,4 @@ +.venv +model +Scratchpad.ipynb +__pycache__ \ No newline at end of file diff --git a/README.md b/README.md index 6a13b4a..56b313a 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,16 @@ # moondream -tiny vision language model + +a tiny vision language model + +## project goals + +Build a high-quality, low-hallucination vision language model small enough to +run on an edge device without a GPU. + +## moondream0 + +Initial prototype built using SigLIP, Phi-1.5, and the LLaVa training dataset. +The model is for research purposes only, and is subject to the Phi and LLaVa +license restrictions. + + diff --git a/assets/demo-1.jpg b/assets/demo-1.jpg new file mode 100644 index 0000000..ed8a252 Binary files /dev/null and b/assets/demo-1.jpg differ diff --git a/moondream/vision_encoder.py b/moondream/vision_encoder.py new file mode 100644 index 0000000..e1b3a0d --- /dev/null +++ b/moondream/vision_encoder.py @@ -0,0 +1,28 @@ +import torch +from PIL import Image +from torchvision.transforms.v2 import ( + Compose, + Resize, + InterpolationMode, + ToImage, + ToDtype, + Normalize, +) + + +class VisionEncoder: + def __init__(self) -> None: + self.model = torch.jit.load("model/vision.pt").to(dtype=torch.float32) + self.preprocess = Compose( + [ + Resize(size=(384, 384), interpolation=InterpolationMode.BICUBIC), + ToImage(), + ToDtype(torch.float32, scale=True), + Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5]), + ] + ) + + def __call__(self, image: Image) -> torch.Tensor: + with torch.no_grad(): + image_vec = self.preprocess(image.convert("RGB")).unsqueeze(0) + return self.model(image_vec) diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..5c53f9e --- /dev/null +++ b/requirements.txt @@ -0,0 +1,4 @@ +torch +pillow +torchvision +transformers \ No newline at end of file