Specify ref for forked torch.hub repos.

Add einops to Studio imports.
Merge branch 'main' into zoedepth
2026-01-11 14:58:11 -05:00 · 2023-10-19 11:02:13 -05:00 · 2023-10-19 10:30:14 -05:00 · 2023-10-19 06:59:43 -07:00 · 2023-10-08 00:10:13 -04:00
7 changed files with 94 additions and 5 deletions
--- a/apps/stable_diffusion/shark_studio_imports.py
+++ b/apps/stable_diffusion/shark_studio_imports.py
@@ -53,6 +53,7 @@ datas += collect_data_files("jsonschema_specifications")
 datas += collect_data_files("cpuinfo")
 datas += collect_data_files("langchain")
 datas += collect_data_files("cv2")
+datas += collect_data_files("einops")
 datas += [
    ("src/utils/resources/prompts.json", "resources"),
    ("src/utils/resources/model_db.json", "resources"),
--- a/apps/stable_diffusion/src/utils/stable_args.py
+++ b/apps/stable_diffusion/src/utils/stable_args.py
@@ -422,7 +422,7 @@ p.add_argument(

 p.add_argument(
    "--use_stencil",
-    choices=["canny", "openpose", "scribble"],
+    choices=["canny", "openpose", "scribble", "zoedepth"],
    help="Enable the stencil feature.",
 )

--- a/apps/stable_diffusion/src/utils/stencils/init.py
+++ b/apps/stable_diffusion/src/utils/stencils/init.py
@@ -1,2 +1,3 @@
 from apps.stable_diffusion.src.utils.stencils.canny import CannyDetector
 from apps.stable_diffusion.src.utils.stencils.openpose import OpenposeDetector
+from apps.stable_diffusion.src.utils.stencils.zoe import ZoeDetector
--- a/apps/stable_diffusion/src/utils/stencils/stencil_utils.py
+++ b/apps/stable_diffusion/src/utils/stencils/stencil_utils.py
@@ -4,6 +4,7 @@ import torch
 from apps.stable_diffusion.src.utils.stencils import (
    CannyDetector,
    OpenposeDetector,
+    ZoeDetector,
 )

 stencil = {}
@@ -117,6 +118,9 @@ def controlnet_hint_conversion(
        case "scribble":
            print("Working with scribble")
            controlnet_hint = hint_scribble(image)
+        case "zoedepth":
+            print("Working with ZoeDepth")
+            controlnet_hint = hint_zoedepth(image)
        case _:
            return None
    controlnet_hint = controlnet_hint_shaping(
@@ -127,7 +131,7 @@ def controlnet_hint_conversion(

 stencil_to_model_id_map = {
    "canny": "lllyasviel/control_v11p_sd15_canny",
-    "depth": "lllyasviel/control_v11p_sd15_depth",
+    "zoedepth": "lllyasviel/control_v11f1p_sd15_depth",
    "hed": "lllyasviel/sd-controlnet-hed",
    "mlsd": "lllyasviel/control_v11p_sd15_mlsd",
    "normal": "lllyasviel/control_v11p_sd15_normalbae",
@@ -184,3 +188,16 @@ def hint_scribble(image: Image.Image):
        detected_map = np.zeros_like(input_image, dtype=np.uint8)
        detected_map[np.min(input_image, axis=2) < 127] = 255
        return detected_map
+
+
+# Stencil 4. Depth (Only Zoe Preprocessing)
+def hint_zoedepth(image: Image.Image):
+    with torch.no_grad():
+        input_image = np.array(image)
+
+        if not "depth" in stencil:
+            stencil["depth"] = ZoeDetector()
+
+        detected_map = stencil["depth"](input_image)
+        detected_map = HWC3(detected_map)
+        return detected_map
--- a/apps/stable_diffusion/src/utils/stencils/zoe/init.py
+++ b/apps/stable_diffusion/src/utils/stencils/zoe/init.py
@@ -0,0 +1,64 @@
+import numpy as np
+import torch
+from pathlib import Path
+import requests
+
+
+from einops import rearrange
+
+remote_model_path = (
+    "https://huggingface.co/lllyasviel/Annotators/resolve/main/ZoeD_M12_N.pt"
+)
+
+
+class ZoeDetector:
+    def __init__(self):
+        cwd = Path.cwd()
+        ckpt_path = Path(cwd, "stencil_annotator")
+        ckpt_path.mkdir(parents=True, exist_ok=True)
+        modelpath = ckpt_path / "ZoeD_M12_N.pt"
+
+        with requests.get(remote_model_path, stream=True) as r:
+            r.raise_for_status()
+            with open(modelpath, "wb") as f:
+                for chunk in r.iter_content(chunk_size=8192):
+                    f.write(chunk)
+
+        midas = torch.hub.load(
+            "gpetters94/MiDaS:master",
+            "DPT_BEiT_L_384",
+            pretrained=False,
+            force_reload=False,
+        )
+        model = torch.hub.load(
+            "monorimet/ZoeDepth:torch_update",
+            "ZoeD_N",
+            pretrained=False,
+            force_reload=False,
+        )
+        model.load_state_dict(
+            torch.load(modelpath, map_location=model.device)["model"]
+        )
+        model.eval()
+        self.model = model
+
+    def __call__(self, input_image):
+        assert input_image.ndim == 3
+        image_depth = input_image
+        with torch.no_grad():
+            image_depth = torch.from_numpy(image_depth).float()
+            image_depth = image_depth / 255.0
+            image_depth = rearrange(image_depth, "h w c -> 1 c h w")
+            depth = self.model.infer(image_depth)
+
+            depth = depth[0, 0].cpu().numpy()
+
+            vmin = np.percentile(depth, 2)
+            vmax = np.percentile(depth, 85)
+
+            depth -= vmin
+            depth /= vmax - vmin
+            depth = 1.0 - depth
+            depth_image = (depth * 255.0).clip(0, 255).astype(np.uint8)
+
+            return depth_image
--- a/apps/stable_diffusion/web/ui/img2img_ui.py
+++ b/apps/stable_diffusion/web/ui/img2img_ui.py
@@ -453,8 +453,13 @@ with gr.Blocks(title="Image-to-Image") as img2img_web:
                            elem_id="stencil_model",
                            label="Stencil model",
                            value="None",
-                            choices=["None", "canny", "openpose", "scribble"],
-                            allow_custom_value=True,
+                            choices=[
+                                "None",
+                                "canny",
+                                "openpose",
+                                "scribble",
+                                "zoedepth",
+                            ],
                        )

                    def show_canvas(choice):
--- a/requirements.txt
+++ b/requirements.txt
@@ -39,8 +39,9 @@ sentencepiece
 py-cpuinfo
 tiktoken # for codegen
 joblib # for langchain
-timm # for MiniGPT4
+timm==0.9.5 # for MiniGPT4
 langchain
+einops # for zoedepth

 # Keep PyInstaller at the end. Sometimes Windows Defender flags it but most folks can continue even if it errors
 pefile
Author	SHA1	Message	Date
Ean Garvey	791553762b	Specify ref for forked torch.hub repos.	2023-10-19 11:02:13 -05:00
Ean Garvey	0b3c4c029d	Add einops to Studio imports.	2023-10-19 10:30:14 -05:00
Ean Garvey	a498ce17bc	Merge branch 'main' into zoedepth	2023-10-19 06:59:43 -07:00
George Petterson	3c66f87a84	Add ZoeDepth	2023-10-08 00:10:13 -04:00