[WebUi] txt2img_ui: Import png metadata (#1147 )

Drop the torch-mlir pin
Seems to work now with top of master
2026-04-20 03:00:34 -04:00 · 2023-03-10 16:26:34 -08:00 · 2023-03-10 15:39:04 -08:00 · 2023-03-10 12:36:37 -08:00 · 2023-03-10 11:53:08 -08:00 · 2023-03-10 23:31:45 +05:30
38 changed files with 2167 additions and 259 deletions
--- a/.flake8
+++ b/.flake8
@@ -0,0 +1,5 @@
+[flake8]
+count = 1
+show-source = 1
+select = E9,F63,F7,F82
+exclude = lit.cfg.py
--- a/.github/workflows/test-models.yml
+++ b/.github/workflows/test-models.yml
@@ -99,11 +99,12 @@ jobs:
      run: |
        # black format check
        black --version
-        black --line-length 79 --check .
+        black --check .
        # stop the build if there are Python syntax errors or undefined names
-        flake8 . --count --select=E9,F63,F7,F82 --show-source --statistics --exclude lit.cfg.py
+        flake8 . --statistics
        # exit-zero treats all errors as warnings. The GitHub editor is 127 chars wide
-        flake8 . --count --exit-zero --max-complexity=10 --max-line-length=127 --statistics --exclude lit.cfg.py
+        flake8 . --isolated --count --exit-zero --max-complexity=10 --max-line-length=127 \
+          --statistics --exclude lit.cfg.py

    - name: Validate Models on CPU
      if: matrix.suite == 'cpu'
@@ -111,7 +112,7 @@ jobs:
        cd $GITHUB_WORKSPACE
        PYTHON=python${{ matrix.python-version }} IMPORTER=1 ./setup_venv.sh
        source shark.venv/bin/activate
-        pytest --forked --benchmark --ci --ci_sha=${SHORT_SHA} --update_tank -k cpu
+        pytest --forked --benchmark --ci --ci_sha=${SHORT_SHA} --update_tank --tank_url="gs://shark_tank/nightly/" -k cpu 
        gsutil cp ./bench_results.csv gs://shark-public/builder/bench_results/${DATE}/bench_results_cpu_${SHORT_SHA}.csv
        gsutil cp gs://shark-public/builder/bench_results/${DATE}/bench_results_cpu_${SHORT_SHA}.csv gs://shark-public/builder/bench_results/latest/bench_results_cpu_latest.csv

@@ -121,7 +122,7 @@ jobs:
        cd $GITHUB_WORKSPACE
        PYTHON=python${{ matrix.python-version }} BENCHMARK=1 IMPORTER=1 ./setup_venv.sh
        source shark.venv/bin/activate
-        pytest --forked --benchmark --ci --ci_sha=${SHORT_SHA} --update_tank -k cuda
+        pytest --forked --benchmark --ci --ci_sha=${SHORT_SHA} --update_tank --tank_url="gs://shark_tank/nightly/" -k cuda
        gsutil cp ./bench_results.csv gs://shark-public/builder/bench_results/${DATE}/bench_results_cuda_${SHORT_SHA}.csv
        gsutil cp gs://shark-public/builder/bench_results/${DATE}/bench_results_cuda_${SHORT_SHA}.csv gs://shark-public/builder/bench_results/latest/bench_results_cuda_latest.csv
        # Disabled due to black image bug
@@ -136,7 +137,7 @@ jobs:
        export DYLD_LIBRARY_PATH=/usr/local/lib/
        echo $PATH
        pip list | grep -E "torch|iree"
-        pytest --ci --ci_sha=${SHORT_SHA} --local_tank_cache="/Volumes/builder/anush/shark_cache" -k vulkan --update_tank
+        pytest --ci --ci_sha=${SHORT_SHA} --local_tank_cache="/Volumes/builder/anush/shark_cache" --tank_url="gs://shark_tank/nightly/" -k vulkan --update_tank

    - name: Validate Vulkan Models (a100)
      if: matrix.suite == 'vulkan' && matrix.os == 'a100'
@@ -144,7 +145,7 @@ jobs:
        cd $GITHUB_WORKSPACE
        PYTHON=python${{ matrix.python-version }} ./setup_venv.sh
        source shark.venv/bin/activate
-        pytest --forked --benchmark --ci --ci_sha=${SHORT_SHA} --update_tank -k vulkan
+        pytest --forked --benchmark --ci --ci_sha=${SHORT_SHA} --update_tank --tank_url="gs://shark_tank/nightly/" -k vulkan
        python build_tools/stable_diffusion_testing.py --device=vulkan

    - name: Validate Vulkan Models (Windows)
--- a/.style.yapf
+++ b/.style.yapf
@@ -1,3 +0,0 @@
-[style]
-  based_on_style = google
-  column_limit = 80
--- a/apps/stable_diffusion/scripts/img2img.py
+++ b/apps/stable_diffusion/scripts/img2img.py
@@ -32,6 +32,11 @@ img2img_obj = None
 config_obj = None
 schedulers = None

+# set initial values of iree_vulkan_target_triple, use_tuned and import_mlir.
+init_iree_vulkan_target_triple = args.iree_vulkan_target_triple
+init_use_tuned = args.use_tuned
+init_import_mlir = args.import_mlir
+

 # Exposed to UI.
 def img2img_inf(
@@ -131,9 +136,9 @@ def img2img_inf(
        args.height = height
        args.width = width
        args.device = device.split("=>", 1)[1].strip()
-        args.iree_vulkan_target_triple = ""
-        args.use_tuned = True
-        args.import_mlir = True
+        args.iree_vulkan_target_triple = init_iree_vulkan_target_triple
+        args.use_tuned = init_use_tuned
+        args.import_mlir = init_import_mlir
        set_init_device_flags()
        model_id = (
            args.hf_model_id
--- a/apps/stable_diffusion/scripts/inpaint.py
+++ b/apps/stable_diffusion/scripts/inpaint.py
@@ -30,6 +30,11 @@ inpaint_obj = None
 config_obj = None
 schedulers = None

+# set initial values of iree_vulkan_target_triple, use_tuned and import_mlir.
+init_iree_vulkan_target_triple = args.iree_vulkan_target_triple
+init_use_tuned = args.use_tuned
+init_import_mlir = args.import_mlir
+

 # Exposed to UI.
 def inpaint_inf(
@@ -38,6 +43,8 @@ def inpaint_inf(
    image_dict,
    height: int,
    width: int,
+    inpaint_full_res: bool,
+    inpaint_full_res_padding: int,
    steps: int,
    guidance_scale: float,
    seed: int,
@@ -106,9 +113,9 @@ def inpaint_inf(
        args.height = height
        args.width = width
        args.device = device.split("=>", 1)[1].strip()
-        args.iree_vulkan_target_triple = ""
-        args.use_tuned = True
-        args.import_mlir = False
+        args.iree_vulkan_target_triple = init_iree_vulkan_target_triple
+        args.use_tuned = init_use_tuned
+        args.import_mlir = init_import_mlir
        set_init_device_flags()
        model_id = (
            args.hf_model_id
@@ -152,6 +159,8 @@ def inpaint_inf(
            batch_size,
            height,
            width,
+            inpaint_full_res,
+            inpaint_full_res_padding,
            steps,
            guidance_scale,
            img_seed,
@@ -232,6 +241,8 @@ if __name__ == "__main__":
            args.batch_size,
            args.height,
            args.width,
+            args.inpaint_full_res,
+            args.inpaint_full_res_padding,
            args.steps,
            args.guidance_scale,
            seed,
--- a/apps/stable_diffusion/scripts/outpaint.py
+++ b/apps/stable_diffusion/scripts/outpaint.py
@@ -30,6 +30,11 @@ outpaint_obj = None
 config_obj = None
 schedulers = None

+# set initial values of iree_vulkan_target_triple, use_tuned and import_mlir.
+init_iree_vulkan_target_triple = args.iree_vulkan_target_triple
+init_use_tuned = args.use_tuned
+init_import_mlir = args.import_mlir
+

 # Exposed to UI.
 def outpaint_inf(
@@ -110,9 +115,9 @@ def outpaint_inf(
        args.height = height
        args.width = width
        args.device = device.split("=>", 1)[1].strip()
-        args.iree_vulkan_target_triple = ""
-        args.use_tuned = True
-        args.import_mlir = False
+        args.iree_vulkan_target_triple = init_iree_vulkan_target_triple
+        args.use_tuned = init_use_tuned
+        args.import_mlir = init_import_mlir
        set_init_device_flags()
        model_id = (
            args.hf_model_id
--- a/apps/stable_diffusion/scripts/txt2img.py
+++ b/apps/stable_diffusion/scripts/txt2img.py
@@ -29,6 +29,11 @@ txt2img_obj = None
 config_obj = None
 schedulers = None

+# set initial values of iree_vulkan_target_triple, use_tuned and import_mlir.
+init_iree_vulkan_target_triple = args.iree_vulkan_target_triple
+init_use_tuned = args.use_tuned
+init_import_mlir = args.import_mlir
+

 # Exposed to UI.
 def txt2img_inf(
@@ -102,9 +107,9 @@ def txt2img_inf(
        args.height = height
        args.width = width
        args.device = device.split("=>", 1)[1].strip()
-        args.iree_vulkan_target_triple = ""
-        args.use_tuned = True
-        args.import_mlir = False
+        args.iree_vulkan_target_triple = init_iree_vulkan_target_triple
+        args.use_tuned = init_use_tuned
+        args.import_mlir = init_import_mlir
        args.img_path = None
        set_init_device_flags()
        model_id = (
--- a/apps/stable_diffusion/src/models/model_wrappers.py
+++ b/apps/stable_diffusion/src/models/model_wrappers.py
@@ -16,6 +16,7 @@ from apps.stable_diffusion.src.utils import (
    fetch_and_update_base_model_id,
    get_path_stem,
    get_extended_name,
+    get_stencil_model_id,
 )


@@ -81,7 +82,8 @@ class SharkifyStableDiffusionModel:
        use_base_vae: bool = False,
        use_tuned: bool = False,
        low_cpu_mem_usage: bool = False,
-        is_inpaint: bool = False
+        is_inpaint: bool = False,
+        use_stencil: str = None
    ):
        self.check_params(max_len, width, height)
        self.max_len = max_len
@@ -118,6 +120,7 @@ class SharkifyStableDiffusionModel:
        self.model_name = self.model_name + "_" + get_path_stem(self.model_id)
        self.low_cpu_mem_usage = low_cpu_mem_usage
        self.is_inpaint = is_inpaint
+        self.use_stencil = get_stencil_model_id(use_stencil)

    def get_extended_name_for_all_model(self, mask_to_fetch):
        model_name = {}
@@ -229,7 +232,7 @@ class SharkifyStableDiffusionModel:
            ):
                super().__init__()
                self.unet = UNet2DConditionModel.from_pretrained(
-                    "takuma104/control_sd15_canny",  # TODO: ADD with model ID
+                    model_id,
                    subfolder="unet",
                    low_cpu_mem_usage=low_cpu_mem_usage,
                )
@@ -277,12 +280,11 @@ class SharkifyStableDiffusionModel:
    def get_control_net(self):
        class StencilControlNetModel(torch.nn.Module):
            def __init__(
-                self, model_id=self.model_id, low_cpu_mem_usage=False
+                self, model_id=self.use_stencil, low_cpu_mem_usage=False
            ):
                super().__init__()
                self.cnet = ControlNetModel.from_pretrained(
-                    "takuma104/control_sd15_canny", # TODO: ADD with model ID
-                    subfolder="controlnet",
+                    model_id,
                    low_cpu_mem_usage=low_cpu_mem_usage,
                )
                self.in_channels = self.cnet.in_channels
@@ -454,7 +456,7 @@ class SharkifyStableDiffusionModel:
        # --  Fetch all vmfbs for the model, if present, else delete the lot.
        need_vae_encode, need_stencil = False, False
        if args.img_path is not None:
-            if args.use_stencil is not None:
+            if self.use_stencil is not None:
                need_stencil = True
            else:
                need_vae_encode = True
--- a/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_inpaint.py
+++ b/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_inpaint.py
@@ -2,7 +2,7 @@ import torch
 from tqdm.auto import tqdm
 import numpy as np
 from random import randint
-from PIL import Image
+from PIL import Image, ImageOps
 from transformers import CLIPTokenizer
 from typing import Union
 from shark.shark_inference import SharkInference
@@ -43,10 +43,222 @@ class InpaintPipeline(StableDiffusionPipeline):
        super().__init__(vae, text_encoder, tokenizer, unet, scheduler)
        self.vae_encode = vae_encode

-    def prepare_mask_and_masked_image(self, image, mask, height, width):
+    def prepare_latents(
+        self,
+        batch_size,
+        height,
+        width,
+        generator,
+        num_inference_steps,
+        dtype,
+    ):
+        latents = torch.randn(
+            (
+                batch_size,
+                4,
+                height // 8,
+                width // 8,
+            ),
+            generator=generator,
+            dtype=torch.float32,
+        ).to(dtype)
+
+        self.scheduler.set_timesteps(num_inference_steps)
+        latents = latents * self.scheduler.init_noise_sigma
+        return latents
+
+    def get_crop_region(self, mask, pad=0):
+        h, w = mask.shape
+
+        crop_left = 0
+        for i in range(w):
+            if not (mask[:, i] == 0).all():
+                break
+            crop_left += 1
+
+        crop_right = 0
+        for i in reversed(range(w)):
+            if not (mask[:, i] == 0).all():
+                break
+            crop_right += 1
+
+        crop_top = 0
+        for i in range(h):
+            if not (mask[i] == 0).all():
+                break
+            crop_top += 1
+
+        crop_bottom = 0
+        for i in reversed(range(h)):
+            if not (mask[i] == 0).all():
+                break
+            crop_bottom += 1
+
+        return (
+            int(max(crop_left - pad, 0)),
+            int(max(crop_top - pad, 0)),
+            int(min(w - crop_right + pad, w)),
+            int(min(h - crop_bottom + pad, h)),
+        )
+
+    def expand_crop_region(
+        self,
+        crop_region,
+        processing_width,
+        processing_height,
+        image_width,
+        image_height,
+    ):
+        x1, y1, x2, y2 = crop_region
+
+        ratio_crop_region = (x2 - x1) / (y2 - y1)
+        ratio_processing = processing_width / processing_height
+
+        if ratio_crop_region > ratio_processing:
+            desired_height = (x2 - x1) / ratio_processing
+            desired_height_diff = int(desired_height - (y2 - y1))
+            y1 -= desired_height_diff // 2
+            y2 += desired_height_diff - desired_height_diff // 2
+            if y2 >= image_height:
+                diff = y2 - image_height
+                y2 -= diff
+                y1 -= diff
+            if y1 < 0:
+                y2 -= y1
+                y1 -= y1
+            if y2 >= image_height:
+                y2 = image_height
+        else:
+            desired_width = (y2 - y1) * ratio_processing
+            desired_width_diff = int(desired_width - (x2 - x1))
+            x1 -= desired_width_diff // 2
+            x2 += desired_width_diff - desired_width_diff // 2
+            if x2 >= image_width:
+                diff = x2 - image_width
+                x2 -= diff
+                x1 -= diff
+            if x1 < 0:
+                x2 -= x1
+                x1 -= x1
+            if x2 >= image_width:
+                x2 = image_width
+
+        return x1, y1, x2, y2
+
+    def resize_image(self, resize_mode, im, width, height):
+        """
+        resize_mode:
+            0: Resize the image to fill the specified width and height, maintaining the aspect ratio, and then center the image within the dimensions, cropping the excess.
+            1: Resize the image to fit within the specified width and height, maintaining the aspect ratio, and then center the image within the dimensions, filling empty with data from image.
+        """
+
+        if resize_mode == 0:
+            ratio = width / height
+            src_ratio = im.width / im.height
+
+            src_w = (
+                width if ratio > src_ratio else im.width * height // im.height
+            )
+            src_h = (
+                height if ratio <= src_ratio else im.height * width // im.width
+            )
+
+            resized = im.resize((src_w, src_h), resample=Image.LANCZOS)
+            res = Image.new("RGB", (width, height))
+            res.paste(
+                resized,
+                box=(width // 2 - src_w // 2, height // 2 - src_h // 2),
+            )
+
+        else:
+            ratio = width / height
+            src_ratio = im.width / im.height
+
+            src_w = (
+                width if ratio < src_ratio else im.width * height // im.height
+            )
+            src_h = (
+                height if ratio >= src_ratio else im.height * width // im.width
+            )
+
+            resized = im.resize((src_w, src_h), resample=Image.LANCZOS)
+            res = Image.new("RGB", (width, height))
+            res.paste(
+                resized,
+                box=(width // 2 - src_w // 2, height // 2 - src_h // 2),
+            )
+
+            if ratio < src_ratio:
+                fill_height = height // 2 - src_h // 2
+                res.paste(
+                    resized.resize((width, fill_height), box=(0, 0, width, 0)),
+                    box=(0, 0),
+                )
+                res.paste(
+                    resized.resize(
+                        (width, fill_height),
+                        box=(0, resized.height, width, resized.height),
+                    ),
+                    box=(0, fill_height + src_h),
+                )
+            elif ratio > src_ratio:
+                fill_width = width // 2 - src_w // 2
+                res.paste(
+                    resized.resize(
+                        (fill_width, height), box=(0, 0, 0, height)
+                    ),
+                    box=(0, 0),
+                )
+                res.paste(
+                    resized.resize(
+                        (fill_width, height),
+                        box=(resized.width, 0, resized.width, height),
+                    ),
+                    box=(fill_width + src_w, 0),
+                )
+
+        return res
+
+    def prepare_mask_and_masked_image(
+        self,
+        image,
+        mask,
+        height,
+        width,
+        inpaint_full_res,
+        inpaint_full_res_padding,
+    ):
        # preprocess image
        image = image.resize((width, height))
        mask = mask.resize((width, height))
+
+        paste_to = ()
+        overlay_image = None
+        if inpaint_full_res:
+            # prepare overlay image
+            overlay_image = Image.new("RGB", (image.width, image.height))
+            overlay_image.paste(
+                image.convert("RGB"),
+                mask=ImageOps.invert(mask.convert("L")),
+            )
+
+            # prepare mask
+            mask = mask.convert("L")
+            crop_region = self.get_crop_region(
+                np.array(mask), inpaint_full_res_padding
+            )
+            crop_region = self.expand_crop_region(
+                crop_region, width, height, mask.width, mask.height
+            )
+            x1, y1, x2, y2 = crop_region
+            mask = mask.crop(crop_region)
+            mask = self.resize_image(1, mask, width, height)
+            paste_to = (x1, y1, x2 - x1, y2 - y1)
+
+            # prepare image
+            image = image.crop(crop_region)
+            image = self.resize_image(1, image, width, height)
+
        if isinstance(image, (Image.Image, np.ndarray)):
            image = [image]

@@ -77,32 +289,7 @@ class InpaintPipeline(StableDiffusionPipeline):

        masked_image = image * (mask < 0.5)

-        return mask, masked_image
-
-    def prepare_latents(
-        self,
-        batch_size,
-        height,
-        width,
-        generator,
-        num_inference_steps,
-        dtype,
-    ):
-        latents = torch.randn(
-            (
-                batch_size,
-                4,
-                height // 8,
-                width // 8,
-            ),
-            generator=generator,
-            dtype=torch.float32,
-        ).to(dtype)
-
-        self.scheduler.set_timesteps(num_inference_steps)
-        self.scheduler.is_scale_input_called = True
-        latents = latents * self.scheduler.init_noise_sigma
-        return latents
+        return mask, masked_image, paste_to, overlay_image

    def prepare_mask_latents(
        self,
@@ -143,6 +330,13 @@ class InpaintPipeline(StableDiffusionPipeline):
            )
        return mask, masked_image_latents

+    def apply_overlay(self, image, paste_loc, overlay):
+        x, y, w, h = paste_loc
+        image = self.resize_image(0, image, w, h)
+        overlay.paste(image, (x, y))
+
+        return overlay
+
    def generate_images(
        self,
        prompts,
@@ -152,6 +346,8 @@ class InpaintPipeline(StableDiffusionPipeline):
        batch_size,
        height,
        width,
+        inpaint_full_res,
+        inpaint_full_res_padding,
        num_inference_steps,
        guidance_scale,
        seed,
@@ -194,8 +390,18 @@ class InpaintPipeline(StableDiffusionPipeline):
        guidance_scale = torch.tensor(guidance_scale).to(torch.float32)

        # Preprocess mask and image
-        mask, masked_image = self.prepare_mask_and_masked_image(
-            image, mask_image, height, width
+        (
+            mask,
+            masked_image,
+            paste_to,
+            overlay_image,
+        ) = self.prepare_mask_and_masked_image(
+            image,
+            mask_image,
+            height,
+            width,
+            inpaint_full_res,
+            inpaint_full_res_padding,
        )

        # Prepare mask latent variables
@@ -230,4 +436,10 @@ class InpaintPipeline(StableDiffusionPipeline):
            )
            all_imgs.extend(imgs)

+        if inpaint_full_res:
+            output_image = self.apply_overlay(
+                all_imgs[0], paste_to, overlay_image
+            )
+            return [output_image]
+
        return all_imgs
--- a/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_outpaint.py
+++ b/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_outpaint.py
@@ -65,7 +65,6 @@ class OutpaintPipeline(StableDiffusionPipeline):
        ).to(dtype)

        self.scheduler.set_timesteps(num_inference_steps)
-        self.scheduler.is_scale_input_called = True
        latents = latents * self.scheduler.init_noise_sigma
        return latents

--- a/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_utils.py
+++ b/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_utils.py
@@ -317,7 +317,7 @@ class StableDiffusionPipeline:
        use_base_vae: bool,
        use_tuned: bool,
        low_cpu_mem_usage: bool = False,
-        use_stencil: bool = False,
+        use_stencil: str = None,
    ):
        is_inpaint = cls.__name__ in [
            "InpaintPipeline",
@@ -337,6 +337,7 @@ class StableDiffusionPipeline:
                use_tuned=use_tuned,
                low_cpu_mem_usage=low_cpu_mem_usage,
                is_inpaint=is_inpaint,
+                use_stencil=use_stencil,
            )
            if cls.__name__ in [
                "Image2ImagePipeline",
--- a/apps/stable_diffusion/src/utils/init.py
+++ b/apps/stable_diffusion/src/utils/init.py
@@ -13,6 +13,7 @@ from apps.stable_diffusion.src.utils.sd_annotation import sd_model_annotation
 from apps.stable_diffusion.src.utils.stable_args import args
 from apps.stable_diffusion.src.utils.stencils.stencil_utils import (
    controlnet_hint_conversion,
+    get_stencil_model_id,
 )
 from apps.stable_diffusion.src.utils.utils import (
    get_shark_model,
--- a/apps/stable_diffusion/src/utils/stable_args.py
+++ b/apps/stable_diffusion/src/utils/stable_args.py
@@ -104,6 +104,21 @@ p.add_argument(
    help="Path to the mask image input for inpainting",
 )

+p.add_argument(
+    "--inpaint_full_res",
+    default=False,
+    action=argparse.BooleanOptionalAction,
+    help="If inpaint only masked area or whole picture",
+)
+
+p.add_argument(
+    "--inpaint_full_res_padding",
+    type=int,
+    default=32,
+    choices=range(0, 257, 4),
+    help="Number of pixels for only masked padding",
+)
+
 p.add_argument(
    "--pixels",
    type=int,
--- a/apps/stable_diffusion/src/utils/stencils/stencil_utils.py
+++ b/apps/stable_diffusion/src/utils/stencils/stencil_utils.py
@@ -134,6 +134,24 @@ def controlnet_hint_conversion(
    return controlnet_hint


+stencil_to_model_id_map = {
+    "canny": "lllyasviel/sd-controlnet-canny",
+    "depth": "lllyasviel/sd-controlnet-depth",
+    "hed": "lllyasviel/sd-controlnet-hed",
+    "mlsd": "lllyasviel/sd-controlnet-mlsd",
+    "normal": "lllyasviel/sd-controlnet-normal",
+    "openpose": "lllyasviel/sd-controlnet-openpose",
+    "scribble": "lllyasviel/sd-controlnet-scribble",
+    "seg": "lllyasviel/sd-controlnet-seg",
+}
+
+
+def get_stencil_model_id(use_stencil):
+    if use_stencil in stencil_to_model_id_map:
+        return stencil_to_model_id_map[use_stencil]
+    return None
+
+
 # Stencil 1. Canny
 def hint_canny(
    image: Image.Image,
--- a/apps/stable_diffusion/src/utils/utils.py
+++ b/apps/stable_diffusion/src/utils/utils.py
@@ -25,11 +25,7 @@ from diffusers.pipelines.stable_diffusion.convert_from_ckpt import (


 def get_extended_name(model_name):
-    device = (
-        args.device
-        if "://" not in args.device
-        else "-".join(args.device.split("://"))
-    )
+    device = args.device.split("://", 1)[0]
    extended_name = "{}_{}".format(model_name, device)
    return extended_name

@@ -603,7 +599,7 @@ def save_output_img(output_img, img_seed, extra_info={}):

    new_entry.update(extra_info)

-    with open(csv_path, "a") as csv_obj:
+    with open(csv_path, "a", encoding="utf-8") as csv_obj:
        dictwriter_obj = DictWriter(csv_obj, fieldnames=list(new_entry.keys()))
        dictwriter_obj.writerow(new_entry)
        csv_obj.close()
--- a/apps/stable_diffusion/web/index.py
+++ b/apps/stable_diffusion/web/index.py
@@ -9,9 +9,12 @@ from apps.stable_diffusion.src import args, clear_all
 from apps.stable_diffusion.web.utils.gradio_configs import (
    clear_gradio_tmp_imgs_folder,
 )
+from apps.stable_diffusion.web.ui.utils import get_custom_model_path

-# clear all gradio tmp images from the last session
+# Clear all gradio tmp images from the last session
 clear_gradio_tmp_imgs_folder()
+# Create the custom model folder if it doesn't already exist
+get_custom_model_path().mkdir(parents=True, exist_ok=True)

 if args.clear_all:
    clear_all()
--- a/apps/stable_diffusion/web/ui/css/sd_dark_theme.css
+++ b/apps/stable_diffusion/web/ui/css/sd_dark_theme.css
@@ -219,3 +219,7 @@ footer {
    pointer-events: none;
 }

+/* Import Png info box */
+#txt2img_prompt_image .fixed-height {
+    height: var(--size-32);
+}
--- a/apps/stable_diffusion/web/ui/img2img_ui.py
+++ b/apps/stable_diffusion/web/ui/img2img_ui.py
@@ -1,6 +1,3 @@
-import os
-import sys
-import glob
 from pathlib import Path
 import gradio as gr
 from PIL import Image
@@ -9,6 +6,10 @@ from apps.stable_diffusion.src import args
 from apps.stable_diffusion.web.ui.utils import (
    available_devices,
    nodlogo_loc,
+    get_custom_model_path,
+    get_custom_model_files,
+    scheduler_list,
+    predefined_models,
 )


@@ -27,32 +28,10 @@ with gr.Blocks(title="Image-to-Image") as img2img_web:
        with gr.Row():
            with gr.Column(scale=1, min_width=600):
                with gr.Row():
-                    ckpt_path = (
-                        Path(args.ckpt_dir)
-                        if args.ckpt_dir
-                        else Path(Path.cwd(), "models")
-                    )
-                    ckpt_path.mkdir(parents=True, exist_ok=True)
-                    types = (
-                        "*.ckpt",
-                        "*.safetensors",
-                    )  # the tuple of file types
-                    ckpt_files = ["None"]
-                    for extn in types:
-                        files = glob.glob(os.path.join(ckpt_path, extn))
-                        ckpt_files.extend(files)
                    custom_model = gr.Dropdown(
-                        label=f"Models (Custom Model path: {ckpt_path})",
+                        label=f"Models (Custom Model path: {get_custom_model_path()})",
                        value=args.ckpt_loc if args.ckpt_loc else "None",
-                        choices=ckpt_files
-                        + [
-                            "Linaqruf/anything-v3.0",
-                            "prompthero/openjourney",
-                            "wavymulder/Analog-Diffusion",
-                            "stabilityai/stable-diffusion-2-1",
-                            "stabilityai/stable-diffusion-2-1-base",
-                            "CompVis/stable-diffusion-v1-4",
-                        ],
+                        choices=get_custom_model_files() + predefined_models,
                    )
                    hf_model_id = gr.Textbox(
                        placeholder="Select 'None' in the Models dropdown on the left and enter model ID here e.g: SG161222/Realistic_Vision_V1.3",
@@ -91,12 +70,7 @@ with gr.Blocks(title="Image-to-Image") as img2img_web:
                        scheduler = gr.Dropdown(
                            label="Scheduler",
                            value="PNDM",
-                            choices=[
-                                "DDIM",
-                                "PNDM",
-                                "DPMSolverMultistep",
-                                "EulerAncestralDiscrete",
-                            ],
+                            choices=scheduler_list,
                        )
                        with gr.Group():
                            save_metadata_to_png = gr.Checkbox(
--- a/apps/stable_diffusion/web/ui/inpaint_ui.py
+++ b/apps/stable_diffusion/web/ui/inpaint_ui.py
@@ -1,6 +1,3 @@
-import os
-import sys
-import glob
 from pathlib import Path
 import gradio as gr
 from PIL import Image
@@ -9,6 +6,10 @@ from apps.stable_diffusion.src import args
 from apps.stable_diffusion.web.ui.utils import (
    available_devices,
    nodlogo_loc,
+    get_custom_model_path,
+    get_custom_model_files,
+    scheduler_list,
+    predefined_paint_models,
 )


@@ -27,28 +28,11 @@ with gr.Blocks(title="Inpainting") as inpaint_web:
        with gr.Row():
            with gr.Column(scale=1, min_width=600):
                with gr.Row():
-                    ckpt_path = (
-                        Path(args.ckpt_dir)
-                        if args.ckpt_dir
-                        else Path(Path.cwd(), "models")
-                    )
-                    ckpt_path.mkdir(parents=True, exist_ok=True)
-                    types = (
-                        "*.ckpt",
-                        "*.safetensors",
-                    )  # the tuple of file types
-                    ckpt_files = ["None"]
-                    for extn in types:
-                        files = glob.glob(os.path.join(ckpt_path, extn))
-                        ckpt_files.extend(files)
                    custom_model = gr.Dropdown(
-                        label=f"Models (Custom Model path: {ckpt_path})",
+                        label=f"Models (Custom Model path: {get_custom_model_path()})",
                        value=args.ckpt_loc if args.ckpt_loc else "None",
-                        choices=ckpt_files
-                        + [
-                            "runwayml/stable-diffusion-inpainting",
-                            "stabilityai/stable-diffusion-2-inpainting",
-                        ],
+                        choices=get_custom_model_files()
+                        + predefined_paint_models,
                    )
                    hf_model_id = gr.Textbox(
                        placeholder="Select 'None' in the Models dropdown on the left and enter model ID here e.g: ghunkins/stable-diffusion-liberty-inpainting",
@@ -83,12 +67,7 @@ with gr.Blocks(title="Inpainting") as inpaint_web:
                        scheduler = gr.Dropdown(
                            label="Scheduler",
                            value="PNDM",
-                            choices=[
-                                "DDIM",
-                                "PNDM",
-                                "DPMSolverMultistep",
-                                "EulerAncestralDiscrete",
-                            ],
+                            choices=scheduler_list,
                        )
                        with gr.Group():
                            save_metadata_to_png = gr.Checkbox(
@@ -126,6 +105,20 @@ with gr.Blocks(title="Inpainting") as inpaint_web:
                            ],
                            visible=False,
                        )
+                    with gr.Row():
+                        inpaint_full_res = gr.Radio(
+                            choices=["Whole picture", "Only masked"],
+                            type="index",
+                            value="Whole picture",
+                            label="Inpaint area",
+                        )
+                        inpaint_full_res_padding = gr.Slider(
+                            minimum=0,
+                            maximum=256,
+                            step=4,
+                            value=32,
+                            label="Only masked padding, pixels",
+                        )
                    with gr.Row():
                        steps = gr.Slider(
                            1, 100, value=args.steps, step=1, label="Steps"
@@ -207,6 +200,8 @@ with gr.Blocks(title="Inpainting") as inpaint_web:
                inpaint_init_image,
                height,
                width,
+                inpaint_full_res,
+                inpaint_full_res_padding,
                steps,
                guidance_scale,
                seed,
--- a/apps/stable_diffusion/web/ui/outpaint_ui.py
+++ b/apps/stable_diffusion/web/ui/outpaint_ui.py
@@ -1,6 +1,3 @@
-import os
-import sys
-import glob
 from pathlib import Path
 import gradio as gr
 from PIL import Image
@@ -9,6 +6,10 @@ from apps.stable_diffusion.src import args
 from apps.stable_diffusion.web.ui.utils import (
    available_devices,
    nodlogo_loc,
+    get_custom_model_path,
+    get_custom_model_files,
+    scheduler_list,
+    predefined_paint_models,
 )


@@ -27,28 +28,11 @@ with gr.Blocks(title="Outpainting") as outpaint_web:
        with gr.Row():
            with gr.Column(scale=1, min_width=600):
                with gr.Row():
-                    ckpt_path = (
-                        Path(args.ckpt_dir)
-                        if args.ckpt_dir
-                        else Path(Path.cwd(), "models")
-                    )
-                    ckpt_path.mkdir(parents=True, exist_ok=True)
-                    types = (
-                        "*.ckpt",
-                        "*.safetensors",
-                    )  # the tuple of file types
-                    ckpt_files = ["None"]
-                    for extn in types:
-                        files = glob.glob(os.path.join(ckpt_path, extn))
-                        ckpt_files.extend(files)
                    custom_model = gr.Dropdown(
-                        label=f"Models (Custom Model path: {ckpt_path})",
+                        label=f"Models (Custom Model path: {get_custom_model_path()})",
                        value=args.ckpt_loc if args.ckpt_loc else "None",
-                        choices=ckpt_files
-                        + [
-                            "runwayml/stable-diffusion-inpainting",
-                            "stabilityai/stable-diffusion-2-inpainting",
-                        ],
+                        choices=get_custom_model_files()
+                        + predefined_paint_models,
                    )
                    hf_model_id = gr.Textbox(
                        placeholder="Select 'None' in the Models dropdown on the left and enter model ID here e.g: ghunkins/stable-diffusion-liberty-inpainting",
@@ -80,12 +64,7 @@ with gr.Blocks(title="Outpainting") as outpaint_web:
                        scheduler = gr.Dropdown(
                            label="Scheduler",
                            value="PNDM",
-                            choices=[
-                                "DDIM",
-                                "PNDM",
-                                "DPMSolverMultistep",
-                                "EulerAncestralDiscrete",
-                            ],
+                            choices=scheduler_list,
                        )
                        with gr.Group():
                            save_metadata_to_png = gr.Checkbox(
--- a/apps/stable_diffusion/web/ui/txt2img_ui.py
+++ b/apps/stable_diffusion/web/ui/txt2img_ui.py
@@ -1,6 +1,3 @@
-import os
-import sys
-import glob
 from pathlib import Path
 import gradio as gr
 from PIL import Image
@@ -9,9 +6,12 @@ from apps.stable_diffusion.src import prompt_examples, args
 from apps.stable_diffusion.web.ui.utils import (
    available_devices,
    nodlogo_loc,
+    get_custom_model_path,
+    get_custom_model_files,
+    scheduler_list_txt2img,
+    predefined_models,
 )

-
 with gr.Blocks(title="Text-to-Image") as txt2img_web:
    with gr.Row(elem_id="ui_title"):
        nod_logo = Image.open(nodlogo_loc)
@@ -27,39 +27,30 @@ with gr.Blocks(title="Text-to-Image") as txt2img_web:
        with gr.Row():
            with gr.Column(scale=1, min_width=600):
                with gr.Row():
-                    ckpt_path = (
-                        Path(args.ckpt_dir)
-                        if args.ckpt_dir
-                        else Path(Path.cwd(), "models")
-                    )
-                    ckpt_path.mkdir(parents=True, exist_ok=True)
-                    types = (
-                        "*.ckpt",
-                        "*.safetensors",
-                    )  # the tuple of file types
-                    ckpt_files = ["None"]
-                    for extn in types:
-                        files = glob.glob(os.path.join(ckpt_path, extn))
-                        ckpt_files.extend(files)
-                    custom_model = gr.Dropdown(
-                        label=f"Models (Custom Model path: {ckpt_path})",
-                        value=args.ckpt_loc if args.ckpt_loc else "None",
-                        choices=ckpt_files
-                        + [
-                            "Linaqruf/anything-v3.0",
-                            "prompthero/openjourney",
-                            "wavymulder/Analog-Diffusion",
-                            "stabilityai/stable-diffusion-2-1",
-                            "stabilityai/stable-diffusion-2-1-base",
-                            "CompVis/stable-diffusion-v1-4",
-                        ],
-                    )
-                    hf_model_id = gr.Textbox(
-                        placeholder="Select 'None' in the Models dropdown on the left and enter model ID here e.g: SG161222/Realistic_Vision_V1.3",
-                        value="",
-                        label="HuggingFace Model ID",
-                        lines=3,
-                    )
+                    with gr.Column(scale=10):
+                        with gr.Row():
+                            custom_model = gr.Dropdown(
+                                label=f"Models (Custom Model path: {get_custom_model_path()})",
+                                value=args.ckpt_loc
+                                if args.ckpt_loc
+                                else "None",
+                                choices=get_custom_model_files()
+                                + predefined_models,
+                            )
+                            hf_model_id = gr.Textbox(
+                                placeholder="Select 'None' in the Models dropdown on the left and enter model ID here e.g: SG161222/Realistic_Vision_V1.3",
+                                value="",
+                                label="HuggingFace Model ID",
+                                lines=3,
+                            )
+                    with gr.Column(scale=1, min_width=170):
+                        png_info_img = gr.Image(
+                            label="Import PNG info",
+                            elem_id="txt2img_prompt_image",
+                            type="pil",
+                            tool="None",
+                            visible=True,
+                        )

                with gr.Group(elem_id="prompt_box_outer"):
                    prompt = gr.Textbox(
@@ -79,16 +70,7 @@ with gr.Blocks(title="Text-to-Image") as txt2img_web:
                        scheduler = gr.Dropdown(
                            label="Scheduler",
                            value=args.scheduler,
-                            choices=[
-                                "DDIM",
-                                "PNDM",
-                                "LMSDiscrete",
-                                "KDPM2Discrete",
-                                "DPMSolverMultistep",
-                                "EulerDiscrete",
-                                "EulerAncestralDiscrete",
-                                "SharkEulerDiscrete",
-                            ],
+                            choices=scheduler_list_txt2img,
                        )
                        with gr.Group():
                            save_metadata_to_png = gr.Checkbox(
@@ -234,3 +216,27 @@ with gr.Blocks(title="Text-to-Image") as txt2img_web:
        prompt.submit(**kwargs)
        negative_prompt.submit(**kwargs)
        stable_diffusion.click(**kwargs)
+
+        from apps.stable_diffusion.web.utils.png_metadata import (
+            import_png_metadata,
+        )
+
+        png_info_img.change(
+            fn=import_png_metadata,
+            inputs=[
+                png_info_img,
+            ],
+            outputs=[
+                png_info_img,
+                prompt,
+                negative_prompt,
+                steps,
+                scheduler,
+                guidance_scale,
+                seed,
+                width,
+                height,
+                custom_model,
+                hf_model_id,
+            ],
+        )
--- a/apps/stable_diffusion/web/ui/utils.py
+++ b/apps/stable_diffusion/web/ui/utils.py
@@ -1,6 +1,44 @@
 import os
 import sys
 from apps.stable_diffusion.src import get_available_devices
+import glob
+from pathlib import Path
+from apps.stable_diffusion.src import args
+
+custom_model_filetypes = (
+    "*.ckpt",
+    "*.safetensors",
+)  # the tuple of file types
+
+scheduler_list = [
+    "DDIM",
+    "PNDM",
+    "DPMSolverMultistep",
+    "EulerAncestralDiscrete",
+]
+scheduler_list_txt2img = [
+    "DDIM",
+    "PNDM",
+    "LMSDiscrete",
+    "KDPM2Discrete",
+    "DPMSolverMultistep",
+    "EulerDiscrete",
+    "EulerAncestralDiscrete",
+    "SharkEulerDiscrete",
+]
+
+predefined_models = [
+    "Linaqruf/anything-v3.0",
+    "prompthero/openjourney",
+    "wavymulder/Analog-Diffusion",
+    "stabilityai/stable-diffusion-2-1",
+    "stabilityai/stable-diffusion-2-1-base",
+    "CompVis/stable-diffusion-v1-4",
+]
+predefined_paint_models = [
+    "runwayml/stable-diffusion-inpainting",
+    "stabilityai/stable-diffusion-2-inpainting",
+]


 def resource_path(relative_path):
@@ -11,5 +49,17 @@ def resource_path(relative_path):
    return os.path.join(base_path, relative_path)


+def get_custom_model_path():
+    return Path(args.ckpt_dir) if args.ckpt_dir else Path(Path.cwd(), "models")
+
+
+def get_custom_model_files():
+    ckpt_files = ["None"]
+    for extn in custom_model_filetypes:
+        files = glob.glob(os.path.join(get_custom_model_path(), extn))
+        ckpt_files.extend(files)
+    return ckpt_files
+
+
 nodlogo_loc = resource_path("logos/nod-logo.png")
 available_devices = get_available_devices()
--- a/apps/stable_diffusion/web/utils/png_metadata.py
+++ b/apps/stable_diffusion/web/utils/png_metadata.py
@@ -0,0 +1,121 @@
+import re
+import os
+from pathlib import Path
+from apps.stable_diffusion.web.ui.txt2img_ui import (
+    png_info_img,
+    prompt,
+    negative_prompt,
+    steps,
+    scheduler,
+    guidance_scale,
+    seed,
+    width,
+    height,
+    custom_model,
+    hf_model_id,
+)
+from apps.stable_diffusion.web.ui.utils import (
+    get_custom_model_path,
+    scheduler_list_txt2img,
+    predefined_models,
+)
+
+re_param_code = r'\s*([\w ]+):\s*("(?:\\"[^,]|\\"|\\|[^\"])+"|[^,]*)(?:,|$)'
+re_param = re.compile(re_param_code)
+re_imagesize = re.compile(r"^(\d+)x(\d+)$")
+
+
+def parse_generation_parameters(x: str):
+    res = {}
+    prompt = ""
+    negative_prompt = ""
+    done_with_prompt = False
+
+    *lines, lastline = x.strip().split("\n")
+    if len(re_param.findall(lastline)) < 3:
+        lines.append(lastline)
+        lastline = ""
+
+    for i, line in enumerate(lines):
+        line = line.strip()
+        if line.startswith("Negative prompt:"):
+            done_with_prompt = True
+            line = line[16:].strip()
+
+        if done_with_prompt:
+            negative_prompt += ("" if negative_prompt == "" else "\n") + line
+        else:
+            prompt += ("" if prompt == "" else "\n") + line
+
+    res["Prompt"] = prompt
+    res["Negative prompt"] = negative_prompt
+
+    for k, v in re_param.findall(lastline):
+        v = v[1:-1] if v[0] == '"' and v[-1] == '"' else v
+        m = re_imagesize.match(v)
+        if m is not None:
+            res[k + "-1"] = m.group(1)
+            res[k + "-2"] = m.group(2)
+        else:
+            res[k] = v
+
+    # Missing CLIP skip means it was set to 1 (the default)
+    if "Clip skip" not in res:
+        res["Clip skip"] = "1"
+
+    hypernet = res.get("Hypernet", None)
+    if hypernet is not None:
+        res[
+            "Prompt"
+        ] += f"""<hypernet:{hypernet}:{res.get("Hypernet strength", "1.0")}>"""
+
+    if "Hires resize-1" not in res:
+        res["Hires resize-1"] = 0
+        res["Hires resize-2"] = 0
+
+    return res
+
+
+def import_png_metadata(pil_data):
+    try:
+        png_info = pil_data.info["parameters"]
+        metadata = parse_generation_parameters(png_info)
+        png_hf_model_id = ""
+
+        # Check for a model match with one of the local ckpt or safetensors files
+        ckpt_path = get_custom_model_path()
+        png_custom_model = os.path.join(ckpt_path, metadata["Model"])
+        if not Path(png_custom_model).is_file():
+            png_custom_model = "None"
+        # Check for a model match with one of the default model list (ex: "Linaqruf/anything-v3.0")
+        if metadata["Model"] in predefined_models:
+            png_custom_model = metadata["Model"]
+        # If nothing was found, fallback to hf model id
+        if png_custom_model == "None":
+            png_hf_model_id = metadata["Model"]
+
+        outputs = {
+            png_info_img: None,
+            negative_prompt: metadata["Negative prompt"],
+            steps: int(metadata["Steps"]),
+            guidance_scale: float(metadata["CFG scale"]),
+            seed: int(metadata["Seed"]),
+            width: float(metadata["Size-1"]),
+            height: float(metadata["Size-2"]),
+            custom_model: png_custom_model,
+            hf_model_id: png_hf_model_id,
+        }
+        if metadata["Prompt"]:
+            outputs[prompt] = metadata["Prompt"]
+        if metadata["Sampler"] in scheduler_list_txt2img:
+            outputs[scheduler] = metadata["Sampler"]
+        return outputs
+
+    except Exception as ex:
+        if pil_data and pil_data.info.get("parameters"):
+            print("import_png_metadata failed with %s" % ex)
+        pass
+
+    return {
+        png_info_img: None,
+    }
--- a/cpp/README.md
+++ b/cpp/README.md
@@ -40,7 +40,7 @@ cmake --build build/
 *Prepare the model*
 ```bash
 wget https://storage.googleapis.com/shark_tank/latest/resnet50_tf/resnet50_tf.mlir
-iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --iree-llvm-embedded-linker-path=`python3 -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])'`/iree/compiler/tools/../_mlir_libs/iree-lld --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --mlir-pass-pipeline-crash-reproducer=ist/core-reproducer.mlir --iree-llvm-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 resnet50_tf.mlir -o resnet50_tf.vmfb
+iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --iree-llvmcpu-embedded-linker-path=`python3 -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])'`/iree/compiler/tools/../_mlir_libs/iree-lld --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --mlir-pass-pipeline-crash-reproducer=ist/core-reproducer.mlir --iree-llvmcpu-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 resnet50_tf.mlir -o resnet50_tf.vmfb
 ```
 *Prepare the input*

@@ -65,18 +65,18 @@ A tool for benchmarking other models is built and can be invoked with a command
 see `./build/vulkan_gui/iree-vulkan-gui --help` for an explanation on the function input. For example, stable diffusion unet can be tested with the following commands:
 ```bash
 wget https://storage.googleapis.com/shark_tank/quinn/stable_diff_tf/stable_diff_tf.mlir
-iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --iree-llvm-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 stable_diff_tf.mlir -o stable_diff_tf.vmfb
+iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --iree-llvmcpu-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 stable_diff_tf.mlir -o stable_diff_tf.vmfb
 ./build/vulkan_gui/iree-vulkan-gui --module-file=stable_diff_tf.vmfb --function_input=2x4x64x64xf32 --function_input=1xf32 --function_input=2x77x768xf32
 ```
 VAE and Autoencoder are also available
 ```bash
 # VAE
 wget https://storage.googleapis.com/shark_tank/quinn/stable_diff_tf/vae_tf/vae.mlir
-iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --iree-llvm-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 vae.mlir -o vae.vmfb
+iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --iree-llvmcpu-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 vae.mlir -o vae.vmfb
 ./build/vulkan_gui/iree-vulkan-gui --module-file=stable_diff_tf.vmfb --function_input=1x4x64x64xf32

 # CLIP Autoencoder
 wget https://storage.googleapis.com/shark_tank/quinn/stable_diff_tf/clip_tf/clip_autoencoder.mlir
-iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --iree-llvm-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 clip_autoencoder.mlir -o clip_autoencoder.vmfb
+iree-compile --iree-input-type=mhlo --iree-vm-bytecode-module-output-format=flatbuffer-binary --iree-hal-target-backends=vulkan --mlir-print-debuginfo --mlir-print-op-on-diagnostic=false --iree-llvmcpu-target-cpu-features=host -iree-vulkan-target-triple=rdna2-unknown-linux --iree-stream-resource-index-bits=64 --iree-vm-target-index-bits=64 clip_autoencoder.mlir -o clip_autoencoder.vmfb
 ./build/vulkan_gui/iree-vulkan-gui --module-file=stable_diff_tf.vmfb --function_input=1x77xi32 --function_input=1x77xi32
 ```
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -10,3 +10,6 @@ requires = [
    "iree-runtime>=20221022.190",
 ]
 build-backend = "setuptools.build_meta"
+
+[tool.black]
+line-length = 79
--- a/setup_venv.sh
+++ b/setup_venv.sh
@@ -78,7 +78,7 @@ $PYTHON -m pip install --upgrade -r "$TD/requirements.txt"
 if [ "$torch_mlir_bin" = true ]; then
  if [[ $(uname -s) = 'Darwin' ]]; then
    echo "MacOS detected. Installing torch-mlir from .whl, to avoid dependency problems with torch."
-    $PYTHON -m pip install --pre --no-cache-dir  torch-mlir -f https://llvm.github.io/torch-mlir/package-index/ -f https://download.pytorch.org/whl/nightly/torch/
+    $PYTHON -m pip install --pre --no-cache-dir torch-mlir -f https://llvm.github.io/torch-mlir/package-index/ -f https://download.pytorch.org/whl/nightly/torch/
  else
    $PYTHON -m pip install --pre torch-mlir -f https://llvm.github.io/torch-mlir/package-index/
    if [ $? -eq 0 ];then
@@ -102,7 +102,7 @@ else
 fi
 if [[ -z "${NO_BACKEND}" ]]; then
  echo "Installing ${RUNTIME}..."
-  $PYTHON -m pip install --upgrade --find-links ${RUNTIME} iree-compiler iree-runtime
+  $PYTHON -m pip install --pre --upgrade --find-links ${RUNTIME} iree-compiler iree-runtime
 else
  echo "Not installing a backend, please make sure to add your backend to PYTHONPATH"
 fi
--- a/shark/examples/shark_inference/llama/README.md
+++ b/shark/examples/shark_inference/llama/README.md
@@ -0,0 +1,18 @@
+# SHARK LLaMA
+
+## TORCH-MLIR Version
+
+```
+https://github.com/nod-ai/torch-mlir.git
+```
+Then check out the `complex` branch and build.
+
+### Setup & Run
+```
+git clone https://github.com/nod-ai/llama.git
+```
+Then in this repository
+```
+pip install -e .
+python llama/shark_model.py
+```
--- a/shark/examples/shark_inference/sharded_bloom.py
+++ b/shark/examples/shark_inference/sharded_bloom.py
@@ -7,9 +7,13 @@
 # -de --device: the device you want to run bloom on.  E.G. cpu, cuda
 # -c, --recompile: set to true if you want to recompile to vmfb.
 # -d, --download: set to true if you want to redownload the mlir files
+# -cm, --create_mlirs: set to true if you want to create the mlir files from scratch.  please make sure you have transformers 4.21.2 before using this option
 # -t --token_count: the number of tokens you want to generate
 # -pr --prompt: the prompt you want to feed to the model
-# -m --model_namme: the name of the model, e.g. bloom-560m
+# -m --model_name: the name of the model, e.g. bloom-560m
+#
+# If you don't specify a prompt when you run this example, you will be able to give prompts through the terminal.  Run the
+# example in this way if you want to run multiple examples without reinitializing the model
 #####################################################################################

 import os
@@ -26,6 +30,7 @@ import sys
 import argparse
 import json
 import urllib.request
+import subprocess

 from torch.fx.experimental.proxy_tensor import make_fx
 from torch._decomp import get_decompositions
@@ -306,31 +311,8 @@ def _prepare_attn_mask(

 def download_model(destination_folder, model_name):
    download_public_file(
-        f"https://{model_name}/config.json", destination_folder
+        f"gs://shark_tank/sharded_bloom/{model_name}/", destination_folder
    )
-    f = open(f"{destination_folder}/config.json")
-    config = json.load(f)
-    f.close()
-    n_blocks = config["n_layer"]
-    download_public_file(
-        f"https://{model_name}/lm_head.mlir", destination_folder
-    )
-    download_public_file(f"https://{model_name}/ln_f.mlir", destination_folder)
-    download_public_file(
-        f"https://{model_name}/word_embeddings.mlir", destination_folder
-    )
-    download_public_file(
-        f"https://{model_name}/word_embeddings_layernorm.mlir",
-        destination_folder,
-    )
-    download_public_file(
-        f"https://{model_name}/tokenizer.json", destination_folder
-    )
-
-    for i in range(n_blocks):
-        download_public_file(
-            f"https://{model_name}/bloom_block_{i}.mlir", destination_folder
-        )


 def compile_embeddings(embeddings_layer, input_ids, path):
@@ -653,6 +635,75 @@ def create_mlirs(destination_folder, model_name):
    )


+def run_large_model(
+    token_count,
+    recompile,
+    model_path,
+    prompt,
+    device_list,
+    script_path,
+    device,
+):
+    f = open(f"{model_path}/prompt.txt", "w+")
+    f.write(prompt)
+    f.close()
+    for i in range(token_count):
+        if i == 0:
+            will_compile = recompile
+        else:
+            will_compile = False
+            f = open(f"{model_path}/prompt.txt", "r")
+            prompt = f.read()
+            f.close()
+
+        subprocess.run(
+            [
+                "python",
+                script_path,
+                model_path,
+                "start",
+                str(will_compile),
+                "cpu",
+                "None",
+                prompt,
+            ]
+        )
+        for i in range(config["n_layer"]):
+            if device_list is not None:
+                device_idx = str(device_list[i % len(device_list)])
+            else:
+                device_idx = "None"
+            subprocess.run(
+                [
+                    "python",
+                    script_path,
+                    model_path,
+                    str(i),
+                    str(will_compile),
+                    device,
+                    device_idx,
+                    prompt,
+                ]
+            )
+        subprocess.run(
+            [
+                "python",
+                script_path,
+                model_path,
+                "end",
+                str(will_compile),
+                "cpu",
+                "None",
+                prompt,
+            ]
+        )
+
+    f = open(f"{model_path}/prompt.txt", "r")
+    output = f.read()
+    f.close()
+    print(output)
+
+
 if __name__ == "__main__":
    parser = argparse.ArgumentParser(prog="Bloom-560m")
    parser.add_argument("-p", "--model_path")
@@ -662,37 +713,130 @@ if __name__ == "__main__":
    parser.add_argument("-d", "--download", default=False, type=bool)
    parser.add_argument("-t", "--token_count", default=10, type=int)
    parser.add_argument("-m", "--model_name", default="bloom-560m")
+    parser.add_argument("-cm", "--create_mlirs", default=False, type=bool)
+
+    parser.add_argument(
+        "-lm", "--large_model_memory_efficient", default=False, type=bool
+    )
+
    parser.add_argument(
        "-pr",
        "--prompt",
-        default="The SQL command to extract all the users whose name starts with A is: ",
+        default=None,
    )
    args = parser.parse_args()

+    if args.create_mlirs and args.large_model_memory_efficient:
+        print(
+            "Warning: If you need to use memory efficient mode, you probably want to use 'download' instead"
+        )
+
+    if not os.path.isdir(args.model_path):
+        os.mkdir(args.model_path)
+
    if args.device_list is not None:
        args.device_list = json.loads(args.device_list)

    if args.device == "cuda" and args.device_list is not None:
        IS_CUDA = True
        from cuda.cudart import cudaSetDevice
+    if args.download and args.create_mlirs:
+        print(
+            "WARNING: It is not advised to turn on both download and create_mlirs"
+        )
    if args.download:
-        # download_model(args.model_path, args.model_name)
+        download_model(args.model_path, args.model_name)
+    if args.create_mlirs:
        create_mlirs(args.model_path, args.model_name)
    from transformers import AutoTokenizer, AutoModelForCausalLM, BloomConfig

    tokenizer = AutoTokenizer.from_pretrained(args.model_path)
-    input_ids = tokenizer.encode(args.prompt, return_tensors="pt")
+    if args.prompt is not None:
+        input_ids = tokenizer.encode(args.prompt, return_tensors="pt")

-    shardedbloom = ShardedBloom(args.model_path)
-    shardedbloom.init_layers(
-        device=args.device, replace=args.recompile, device_idx=args.device_list
-    )
-    shardedbloom.load_layers()
+    if args.large_model_memory_efficient:
+        f = open(f"{args.model_path}/config.json")
+        config = json.load(f)
+        f.close()

-    for _ in range(args.token_count):
-        next_token = shardedbloom.forward_pass(
-            torch.tensor(input_ids), device=args.device
+        self_path = os.path.dirname(os.path.abspath(__file__))
+        script_path = os.path.join(self_path, "sharded_bloom_large_models.py")
+
+        if args.prompt is not None:
+            run_large_model(
+                args.token_count,
+                args.recompile,
+                args.model_path,
+                args.prompt,
+                args.device_list,
+                script_path,
+                args.device,
+            )
+
+        else:
+            while True:
+                prompt = input("Enter Prompt: ")
+                try:
+                    token_count = int(
+                        input("Enter number of tokens you want to generate: ")
+                    )
+                except:
+                    print(
+                        "Invalid integer entered.  Using default value of 10"
+                    )
+                    token_count = 10
+
+                run_large_model(
+                    token_count,
+                    args.recompile,
+                    args.model_path,
+                    prompt,
+                    args.device_list,
+                    script_path,
+                    args.device,
+                )
+
+    else:
+        shardedbloom = ShardedBloom(args.model_path)
+        shardedbloom.init_layers(
+            device=args.device,
+            replace=args.recompile,
+            device_idx=args.device_list,
        )
-        input_ids = torch.cat([input_ids, next_token.unsqueeze(-1)], dim=-1)
+        shardedbloom.load_layers()

-    print(tokenizer.decode(input_ids.squeeze()))
+        if args.prompt is not None:
+            for _ in range(args.token_count):
+                next_token = shardedbloom.forward_pass(
+                    torch.tensor(input_ids), device=args.device
+                )
+                input_ids = torch.cat(
+                    [input_ids, next_token.unsqueeze(-1)], dim=-1
+                )
+
+            print(tokenizer.decode(input_ids.squeeze()))
+
+        else:
+            while True:
+                prompt = input("Enter Prompt: ")
+                try:
+                    token_count = int(
+                        input("Enter number of tokens you want to generate: ")
+                    )
+                except:
+                    print(
+                        "Invalid integer entered.  Using default value of 10"
+                    )
+                    token_count = 10
+
+                input_ids = tokenizer.encode(prompt, return_tensors="pt")
+
+                for _ in range(token_count):
+                    next_token = shardedbloom.forward_pass(
+                        torch.tensor(input_ids), device=args.device
+                    )
+                    input_ids = torch.cat(
+                        [input_ids, next_token.unsqueeze(-1)], dim=-1
+                    )
+
+                print(tokenizer.decode(input_ids.squeeze()))
--- a/shark/examples/shark_inference/sharded_bloom_large_models.py
+++ b/shark/examples/shark_inference/sharded_bloom_large_models.py
@@ -0,0 +1,381 @@
+import sys
+import os
+from transformers import AutoTokenizer, AutoModelForCausalLM, BloomConfig
+import re
+from shark.shark_inference import SharkInference
+import torch
+import torch.nn as nn
+from collections import OrderedDict
+from transformers.models.bloom.modeling_bloom import (
+    BloomBlock,
+    build_alibi_tensor,
+)
+import time
+import json
+
+
+def _expand_mask(mask: torch.Tensor, dtype: torch.dtype, tgt_len: int = None):
+    """
+    Expands attention_mask from `[bsz, seq_len]` to `[bsz, 1, tgt_seq_len, src_seq_len]`.
+    """
+    batch_size, source_length = mask.size()
+    tgt_len = tgt_len if tgt_len is not None else source_length
+
+    expanded_mask = (
+        mask[:, None, None, :]
+        .expand(batch_size, 1, tgt_len, source_length)
+        .to(dtype)
+    )
+
+    inverted_mask = 1.0 - expanded_mask
+
+    return inverted_mask.masked_fill(
+        inverted_mask.to(torch.bool), torch.finfo(dtype).min
+    )
+
+
+def _prepare_attn_mask(
+    attention_mask, input_shape, inputs_embeds, past_key_values_length
+):
+    # create causal mask
+    # [bsz, seq_len] -> [bsz, 1, tgt_seq_len, src_seq_len]
+    combined_attention_mask = None
+    if input_shape[-1] > 1:
+        combined_attention_mask = _make_causal_mask(
+            input_shape,
+            inputs_embeds.dtype,
+            past_key_values_length=past_key_values_length,
+        ).to(attention_mask.device)
+
+    if attention_mask is not None:
+        # [bsz, seq_len] -> [bsz, 1, tgt_seq_len, src_seq_len]
+        expanded_attn_mask = _expand_mask(
+            attention_mask, inputs_embeds.dtype, tgt_len=input_shape[-1]
+        )
+        combined_attention_mask = (
+            expanded_attn_mask
+            if combined_attention_mask is None
+            else expanded_attn_mask + combined_attention_mask
+        )
+
+    return combined_attention_mask
+
+
+def _make_causal_mask(
+    input_ids_shape: torch.Size,
+    dtype: torch.dtype,
+    past_key_values_length: int = 0,
+):
+    """
+    Make causal mask used for bi-directional self-attention.
+    """
+    batch_size, target_length = input_ids_shape
+    mask = torch.full((target_length, target_length), torch.finfo(dtype).min)
+    mask_cond = torch.arange(mask.size(-1))
+    intermediate_mask = mask_cond < (mask_cond + 1).view(mask.size(-1), 1)
+    mask.masked_fill_(intermediate_mask, 0)
+    mask = mask.to(dtype)
+
+    if past_key_values_length > 0:
+        mask = torch.cat(
+            [
+                torch.zeros(
+                    target_length, past_key_values_length, dtype=dtype
+                ),
+                mask,
+            ],
+            dim=-1,
+        )
+    expanded_mask = mask[None, None, :, :].expand(
+        batch_size, 1, target_length, target_length + past_key_values_length
+    )
+    return expanded_mask
+
+
+if __name__ == "__main__":
+    working_dir = sys.argv[1]
+    layer_name = sys.argv[2]
+    will_compile = sys.argv[3]
+    device = sys.argv[4]
+    device_idx = sys.argv[5]
+    prompt = sys.argv[6]
+
+    if device_idx.lower().strip() == "none":
+        device_idx = None
+    else:
+        device_idx = int(device_idx)
+
+    if will_compile.lower().strip() == "true":
+        will_compile = True
+    else:
+        will_compile = False
+
+    f = open(f"{working_dir}/config.json")
+    config = json.load(f)
+    f.close()
+
+    layers_initialized = False
+    try:
+        n_embed = config["n_embed"]
+    except KeyError:
+        n_embed = config["hidden_size"]
+    vocab_size = config["vocab_size"]
+    n_layer = config["n_layer"]
+    try:
+        n_head = config["num_attention_heads"]
+    except KeyError:
+        n_head = config["n_head"]
+
+    if not os.path.isdir(working_dir):
+        os.mkdir(working_dir)
+
+    if layer_name == "start":
+        tokenizer = AutoTokenizer.from_pretrained(working_dir)
+        input_ids = tokenizer.encode(prompt, return_tensors="pt")
+
+        mlir_str = ""
+
+        if will_compile:
+            f = open(f"{working_dir}/word_embeddings.mlir", encoding="utf-8")
+            mlir_str = f.read()
+            f.close()
+
+            mlir_str = bytes(mlir_str, "utf-8")
+
+        shark_module = SharkInference(
+            mlir_str,
+            device="cpu",
+            mlir_dialect="tm_tensor",
+            device_idx=None,
+        )
+
+        if will_compile:
+            shark_module.save_module(
+                module_name=f"{working_dir}/word_embeddings",
+                extra_args=[
+                    "--iree-vm-bytecode-module-output-format=flatbuffer-binary",
+                    "--iree-stream-resource-max-allocation-size=1000000000",
+                    "--iree-codegen-check-ir-before-llvm-conversion=false",
+                ],
+            )
+
+        shark_module.load_module(f"{working_dir}/word_embeddings.vmfb")
+        input_embeds = shark_module(
+            inputs=(input_ids,), function_name="forward"
+        )
+        input_embeds = torch.tensor(input_embeds).float()
+
+        mlir_str = ""
+
+        if will_compile:
+            f = open(
+                f"{working_dir}/word_embeddings_layernorm.mlir",
+                encoding="utf-8",
+            )
+            mlir_str = f.read()
+            f.close()
+
+        shark_module = SharkInference(
+            mlir_str,
+            device="cpu",
+            mlir_dialect="tm_tensor",
+            device_idx=None,
+        )
+
+        if will_compile:
+            shark_module.save_module(
+                module_name=f"{working_dir}/word_embeddings_layernorm",
+                extra_args=[
+                    "--iree-vm-bytecode-module-output-format=flatbuffer-binary",
+                    "--iree-stream-resource-max-allocation-size=1000000000",
+                    "--iree-codegen-check-ir-before-llvm-conversion=false",
+                ],
+            )
+
+        shark_module.load_module(
+            f"{working_dir}/word_embeddings_layernorm.vmfb"
+        )
+        hidden_states = shark_module(
+            inputs=(input_embeds,), function_name="forward"
+        )
+        hidden_states = torch.tensor(hidden_states).float()
+
+        torch.save(hidden_states, f"{working_dir}/hidden_states_0.pt")
+
+        attention_mask = torch.ones(
+            [hidden_states.shape[0], len(input_ids[0])]
+        )
+
+        attention_mask = torch.tensor(attention_mask).float()
+
+        alibi = build_alibi_tensor(
+            attention_mask,
+            n_head,
+            hidden_states.dtype,
+            device="cpu",
+        )
+
+        torch.save(alibi, f"{working_dir}/alibi.pt")
+
+        causal_mask = _prepare_attn_mask(
+            attention_mask, input_ids.size(), input_embeds, 0
+        )
+        causal_mask = torch.tensor(causal_mask).float()
+
+        torch.save(causal_mask, f"{working_dir}/causal_mask.pt")
+
+    elif layer_name in [str(x) for x in range(n_layer)]:
+        hidden_states = torch.load(
+            f"{working_dir}/hidden_states_{layer_name}.pt"
+        )
+        alibi = torch.load(f"{working_dir}/alibi.pt")
+        causal_mask = torch.load(f"{working_dir}/causal_mask.pt")
+
+        mlir_str = ""
+
+        if will_compile:
+            f = open(
+                f"{working_dir}/bloom_block_{layer_name}.mlir",
+                encoding="utf-8",
+            )
+            mlir_str = f.read()
+            f.close()
+
+            mlir_str = bytes(mlir_str, "utf-8")
+
+        shark_module = SharkInference(
+            mlir_str,
+            device=device,
+            mlir_dialect="tm_tensor",
+            device_idx=device_idx,
+        )
+
+        if will_compile:
+            shark_module.save_module(
+                module_name=f"{working_dir}/bloom_block_{layer_name}",
+                extra_args=[
+                    "--iree-vm-bytecode-module-output-format=flatbuffer-binary",
+                    "--iree-stream-resource-max-allocation-size=1000000000",
+                    "--iree-codegen-check-ir-before-llvm-conversion=false",
+                ],
+            )
+
+        shark_module.load_module(
+            f"{working_dir}/bloom_block_{layer_name}.vmfb"
+        )
+
+        output = shark_module(
+            inputs=(
+                hidden_states.detach().numpy(),
+                alibi.detach().numpy(),
+                causal_mask.detach().numpy(),
+            ),
+            function_name="forward",
+        )
+
+        hidden_states = torch.tensor(output[0]).float()
+
+        torch.save(
+            hidden_states,
+            f"{working_dir}/hidden_states_{int(layer_name) + 1}.pt",
+        )
+
+    elif layer_name == "end":
+        mlir_str = ""
+
+        if will_compile:
+            f = open(f"{working_dir}/ln_f.mlir", encoding="utf-8")
+            mlir_str = f.read()
+            f.close()
+
+            mlir_str = bytes(mlir_str, "utf-8")
+
+        shark_module = SharkInference(
+            mlir_str,
+            device="cpu",
+            mlir_dialect="tm_tensor",
+            device_idx=None,
+        )
+
+        if will_compile:
+            shark_module.save_module(
+                module_name=f"{working_dir}/ln_f",
+                extra_args=[
+                    "--iree-vm-bytecode-module-output-format=flatbuffer-binary",
+                    "--iree-stream-resource-max-allocation-size=1000000000",
+                    "--iree-codegen-check-ir-before-llvm-conversion=false",
+                ],
+            )
+
+        shark_module.load_module(f"{working_dir}/ln_f.vmfb")
+
+        hidden_states = torch.load(f"{working_dir}/hidden_states_{n_layer}.pt")
+
+        hidden_states = shark_module(
+            inputs=(hidden_states,), function_name="forward"
+        )
+
+        mlir_str = ""
+
+        if will_compile:
+            f = open(f"{working_dir}/lm_head.mlir", encoding="utf-8")
+            mlir_str = f.read()
+            f.close()
+
+            mlir_str = bytes(mlir_str, "utf-8")
+
+        if config["n_embed"] == 14336:
+
+            def get_state_dict():
+                d = torch.load(
+                    f"{working_dir}/pytorch_model_00001-of-00072.bin"
+                )
+                return OrderedDict(
+                    (k.replace("word_embeddings.", ""), v)
+                    for k, v in d.items()
+                )
+
+            def load_causal_lm_head():
+                linear = nn.utils.skip_init(
+                    nn.Linear, 14336, 250880, bias=False, dtype=torch.float
+                )
+                linear.load_state_dict(get_state_dict(), strict=False)
+                return linear.float()
+
+            lm_head = load_causal_lm_head()
+
+            logits = lm_head(torch.tensor(hidden_states).float())
+
+        else:
+            shark_module = SharkInference(
+                mlir_str,
+                device="cpu",
+                mlir_dialect="tm_tensor",
+                device_idx=None,
+            )
+
+            if will_compile:
+                shark_module.save_module(
+                    module_name=f"{working_dir}/lm_head",
+                    extra_args=[
+                        "--iree-vm-bytecode-module-output-format=flatbuffer-binary",
+                        "--iree-stream-resource-max-allocation-size=1000000000",
+                        "--iree-codegen-check-ir-before-llvm-conversion=false",
+                    ],
+                )
+
+            shark_module.load_module(f"{working_dir}/lm_head.vmfb")
+
+            logits = shark_module(
+                inputs=(hidden_states,), function_name="forward"
+            )
+
+        logits = torch.tensor(logits).float()
+
+        tokenizer = AutoTokenizer.from_pretrained(working_dir)
+
+        next_token = tokenizer.decode(torch.argmax(logits[:, -1, :], dim=-1))
+
+        f = open(f"{working_dir}/prompt.txt", "w+")
+        f.write(prompt + next_token)
+        f.close()
--- a/shark/examples/shark_training/stable_diffusion/README.md
+++ b/shark/examples/shark_training/stable_diffusion/README.md
@@ -0,0 +1,43 @@
+# Stable Diffusion Fine Tuning
+
+## Installation (Linux)
+
+### Activate shark.venv Virtual Environment
+
+```shell
+source shark.venv/bin/activate
+
+# Some older pip installs may not be able to handle the recent PyTorch deps
+python -m pip install --upgrade pip
+```
+
+## Install dependencies
+
+### Run the following installation commands:
+```
+pip install -U git+https://github.com/huggingface/diffusers.git
+pip install accelerate transformers ftfy
+```
+
+### Build torch-mlir with the following branch:
+
+Please cherry-pick this branch of torch-mlir: https://github.com/vivekkhandelwal1/torch-mlir/tree/sd-ops
+and build it locally. You can find the instructions for using locally build Torch-MLIR,
+here: https://github.com/nod-ai/SHARK#how-to-use-your-locally-built-iree--torch-mlir-with-shark
+
+## Run the Stable diffusion fine tuning
+
+To run the model with the default set of images and params, run:
+```shell
+python stable_diffusion_fine_tuning.py
+```
+By default the training is run through the PyTorch path. If you want to train the model using the Torchdynamo path of Torch-MLIR, you need to specify `--use_torchdynamo=True`.
+
+The default number of training steps are `2000`, which would take many hours to complete based on your system config. You can pass the smaller value with the arg `--training_steps`. You can specify the number of images to be sampled for the result with the `--num_inference_samples` arg. For the number of inference steps you can use `--inference_steps` flag.
+
+For example, you can run the training for a limited set of steps via the dynamo path by using the following command:
+```
+python stable_diffusion_fine_tuning.py --training_steps=1 --inference_steps=1 --num_inference_samples=1 --train_batch_size=1 --use_torchdynamo=True
+```
+
+You can also specify the device to be used via the flag `--device`. The default value is `cpu`, for GPU execution you can specify `--device="cuda"`.
--- a/shark/examples/shark_training/stable_diffusion/stable_diffusion_fine_tuning.py
+++ b/shark/examples/shark_training/stable_diffusion/stable_diffusion_fine_tuning.py
@@ -0,0 +1,914 @@
+# Install the required libs
+# pip install -U git+https://github.com/huggingface/diffusers.git
+# pip install accelerate transformers ftfy
+
+# Import required libraries
+import argparse
+import itertools
+import math
+import os
+from typing import List
+import random
+
+import numpy as np
+import torch
+import torch.nn.functional as F
+import torch.utils.checkpoint
+from torch.utils.data import Dataset
+
+import PIL
+import logging
+
+import torch_mlir
+from torch_mlir.dynamo import make_simple_dynamo_backend
+import torch._dynamo as dynamo
+from torch.fx.experimental.proxy_tensor import make_fx
+from torch_mlir_e2e_test.linalg_on_tensors_backends import refbackend
+from shark.shark_inference import SharkInference
+
+torch._dynamo.config.verbose = True
+
+from diffusers import (
+    AutoencoderKL,
+    DDPMScheduler,
+    PNDMScheduler,
+    StableDiffusionPipeline,
+    UNet2DConditionModel,
+)
+from diffusers.optimization import get_scheduler
+from diffusers.pipelines.stable_diffusion import (
+    StableDiffusionSafetyChecker,
+)
+from PIL import Image
+from torchvision import transforms
+from tqdm.auto import tqdm
+from transformers import (
+    CLIPFeatureExtractor,
+    CLIPTextModel,
+    CLIPTokenizer,
+)
+
+
+# Enter your HuggingFace Token
+# Note: You can comment this prompt and just set your token instead of passing it through cli for every execution.
+hf_token = input("Please enter your huggingface token here: ")
+YOUR_TOKEN = hf_token
+
+
+def image_grid(imgs, rows, cols):
+    assert len(imgs) == rows * cols
+
+    w, h = imgs[0].size
+    grid = Image.new("RGB", size=(cols * w, rows * h))
+    grid_w, grid_h = grid.size
+
+    for i, img in enumerate(imgs):
+        grid.paste(img, box=(i % cols * w, i // cols * h))
+    return grid
+
+
+# `pretrained_model_name_or_path` which Stable Diffusion checkpoint you want to use
+# Options: 1.) "stabilityai/stable-diffusion-2"
+#          2.) "stabilityai/stable-diffusion-2-base"
+#          3.) "CompVis/stable-diffusion-v1-4"
+#          4.) "runwayml/stable-diffusion-v1-5"
+pretrained_model_name_or_path = "stabilityai/stable-diffusion-2"
+
+# Add here the URLs to the images of the concept you are adding. 3-5 should be fine
+urls = [
+    "https://huggingface.co/datasets/valhalla/images/resolve/main/2.jpeg",
+    "https://huggingface.co/datasets/valhalla/images/resolve/main/3.jpeg",
+    "https://huggingface.co/datasets/valhalla/images/resolve/main/5.jpeg",
+    "https://huggingface.co/datasets/valhalla/images/resolve/main/6.jpeg",
+    ## You can add additional images here
+]
+
+# Downloading Images
+import requests
+import glob
+from io import BytesIO
+
+
+def download_image(url):
+    try:
+        response = requests.get(url)
+    except:
+        return None
+    return Image.open(BytesIO(response.content)).convert("RGB")
+
+
+images = list(filter(None, [download_image(url) for url in urls]))
+save_path = "./my_concept"
+if not os.path.exists(save_path):
+    os.mkdir(save_path)
+[image.save(f"{save_path}/{i}.jpeg") for i, image in enumerate(images)]
+
+p = argparse.ArgumentParser(
+    description=__doc__,
+    formatter_class=argparse.ArgumentDefaultsHelpFormatter,
+)
+p.add_argument(
+    "--input_dir",
+    type=str,
+    default="my_concept/",
+    help="the directory contains the images used for fine tuning",
+)
+p.add_argument(
+    "--output_dir",
+    type=str,
+    default="sd_result",
+    help="the directory contains the images used for fine tuning",
+)
+p.add_argument(
+    "--training_steps",
+    type=int,
+    default=2000,
+    help="the maximum number of training steps",
+)
+p.add_argument(
+    "--train_batch_size",
+    type=int,
+    default=4,
+    help="The batch size for training",
+)
+p.add_argument(
+    "--save_steps",
+    type=int,
+    default=250,
+    help="the number of steps after which to save the learned concept",
+)
+p.add_argument("--seed", type=int, default=42, help="the random seed")
+p.add_argument(
+    "--what_to_teach",
+    type=str,
+    choices=["object", "style"],
+    default="object",
+    help="what is it that you are teaching?",
+)
+p.add_argument(
+    "--placeholder_token",
+    type=str,
+    default="<cat-toy>",
+    help="It is the token you are going to use to represent your new concept",
+)
+p.add_argument(
+    "--initializer_token",
+    type=str,
+    default="toy",
+    help="It is a word that can summarise what is your new concept",
+)
+p.add_argument(
+    "--inference_steps",
+    type=int,
+    default=50,
+    help="the number of steps for inference",
+)
+p.add_argument(
+    "--num_inference_samples",
+    type=int,
+    default=4,
+    help="the number of samples for inference",
+)
+p.add_argument(
+    "--prompt",
+    type=str,
+    default="a grafitti in a wall with a *s on it",
+    help="the text prompt to use",
+)
+p.add_argument(
+    "--device",
+    type=str,
+    default="cpu",
+    help="The device to use",
+)
+p.add_argument(
+    "--use_torchdynamo",
+    type=bool,
+    default=False,
+    help="This flag is used to determine whether the training has to be done through the torchdynamo path or not.",
+)
+args = p.parse_args()
+torch.manual_seed(args.seed)
+
+if "*s" not in args.prompt:
+    raise ValueError(
+        f'The prompt should have a "*s" which will be replaced by a placeholder token.'
+    )
+
+prompt1, prompt2 = args.prompt.split("*s")
+args.prompt = prompt1 + args.placeholder_token + prompt2
+
+# `images_path` is a path to directory containing the training images.
+images_path = args.input_dir
+while not os.path.exists(str(images_path)):
+    print(
+        "The images_path specified does not exist, use the colab file explorer to copy the path :"
+    )
+    images_path = input("")
+save_path = images_path
+
+# Setup and check the images you have just added
+images = []
+for file_path in os.listdir(save_path):
+    try:
+        image_path = os.path.join(save_path, file_path)
+        images.append(Image.open(image_path).resize((512, 512)))
+    except:
+        print(
+            f"{image_path} is not a valid image, please make sure to remove this file from the directory otherwise the training could fail."
+        )
+image_grid(images, 1, len(images))
+
+########### Create Dataset ##########
+
+# Setup the prompt templates for training
+imagenet_templates_small = [
+    "a photo of a {}",
+    "a rendering of a {}",
+    "a cropped photo of the {}",
+    "the photo of a {}",
+    "a photo of a clean {}",
+    "a photo of a dirty {}",
+    "a dark photo of the {}",
+    "a photo of my {}",
+    "a photo of the cool {}",
+    "a close-up photo of a {}",
+    "a bright photo of the {}",
+    "a cropped photo of a {}",
+    "a photo of the {}",
+    "a good photo of the {}",
+    "a photo of one {}",
+    "a close-up photo of the {}",
+    "a rendition of the {}",
+    "a photo of the clean {}",
+    "a rendition of a {}",
+    "a photo of a nice {}",
+    "a good photo of a {}",
+    "a photo of the nice {}",
+    "a photo of the small {}",
+    "a photo of the weird {}",
+    "a photo of the large {}",
+    "a photo of a cool {}",
+    "a photo of a small {}",
+]
+
+imagenet_style_templates_small = [
+    "a painting in the style of {}",
+    "a rendering in the style of {}",
+    "a cropped painting in the style of {}",
+    "the painting in the style of {}",
+    "a clean painting in the style of {}",
+    "a dirty painting in the style of {}",
+    "a dark painting in the style of {}",
+    "a picture in the style of {}",
+    "a cool painting in the style of {}",
+    "a close-up painting in the style of {}",
+    "a bright painting in the style of {}",
+    "a cropped painting in the style of {}",
+    "a good painting in the style of {}",
+    "a close-up painting in the style of {}",
+    "a rendition in the style of {}",
+    "a nice painting in the style of {}",
+    "a small painting in the style of {}",
+    "a weird painting in the style of {}",
+    "a large painting in the style of {}",
+]
+
+
+# Setup the dataset
+class TextualInversionDataset(Dataset):
+    def __init__(
+        self,
+        data_root,
+        tokenizer,
+        learnable_property="object",  # [object, style]
+        size=512,
+        repeats=100,
+        interpolation="bicubic",
+        flip_p=0.5,
+        set="train",
+        placeholder_token="*",
+        center_crop=False,
+    ):
+        self.data_root = data_root
+        self.tokenizer = tokenizer
+        self.learnable_property = learnable_property
+        self.size = size
+        self.placeholder_token = placeholder_token
+        self.center_crop = center_crop
+        self.flip_p = flip_p
+
+        self.image_paths = [
+            os.path.join(self.data_root, file_path)
+            for file_path in os.listdir(self.data_root)
+        ]
+
+        self.num_images = len(self.image_paths)
+        self._length = self.num_images
+
+        if set == "train":
+            self._length = self.num_images * repeats
+
+        self.interpolation = {
+            "linear": PIL.Image.LINEAR,
+            "bilinear": PIL.Image.BILINEAR,
+            "bicubic": PIL.Image.BICUBIC,
+            "lanczos": PIL.Image.LANCZOS,
+        }[interpolation]
+
+        self.templates = (
+            imagenet_style_templates_small
+            if learnable_property == "style"
+            else imagenet_templates_small
+        )
+        self.flip_transform = transforms.RandomHorizontalFlip(p=self.flip_p)
+
+    def __len__(self):
+        return self._length
+
+    def __getitem__(self, i):
+        example = {}
+        image = Image.open(self.image_paths[i % self.num_images])
+
+        if not image.mode == "RGB":
+            image = image.convert("RGB")
+
+        placeholder_string = self.placeholder_token
+        text = random.choice(self.templates).format(placeholder_string)
+
+        example["input_ids"] = self.tokenizer(
+            text,
+            padding="max_length",
+            truncation=True,
+            max_length=self.tokenizer.model_max_length,
+            return_tensors="pt",
+        ).input_ids[0]
+
+        # default to score-sde preprocessing
+        img = np.array(image).astype(np.uint8)
+
+        if self.center_crop:
+            crop = min(img.shape[0], img.shape[1])
+            (
+                h,
+                w,
+            ) = (
+                img.shape[0],
+                img.shape[1],
+            )
+            img = img[
+                (h - crop) // 2 : (h + crop) // 2,
+                (w - crop) // 2 : (w + crop) // 2,
+            ]
+
+        image = Image.fromarray(img)
+        image = image.resize(
+            (self.size, self.size), resample=self.interpolation
+        )
+
+        image = self.flip_transform(image)
+        image = np.array(image).astype(np.uint8)
+        image = (image / 127.5 - 1.0).astype(np.float32)
+
+        example["pixel_values"] = torch.from_numpy(image).permute(2, 0, 1)
+        return example
+
+
+########## Setting up the model ##########
+
+# Load the tokenizer and add the placeholder token as a additional special token.
+tokenizer = CLIPTokenizer.from_pretrained(
+    pretrained_model_name_or_path,
+    subfolder="tokenizer",
+)
+
+# Add the placeholder token in tokenizer
+num_added_tokens = tokenizer.add_tokens(args.placeholder_token)
+if num_added_tokens == 0:
+    raise ValueError(
+        f"The tokenizer already contains the token {args.placeholder_token}. Please pass a different"
+        " `placeholder_token` that is not already in the tokenizer."
+    )
+
+# Get token ids for our placeholder and initializer token.
+# This code block will complain if initializer string is not a single token
+# Convert the initializer_token, placeholder_token to ids
+token_ids = tokenizer.encode(args.initializer_token, add_special_tokens=False)
+# Check if initializer_token is a single token or a sequence of tokens
+if len(token_ids) > 1:
+    raise ValueError("The initializer token must be a single token.")
+
+initializer_token_id = token_ids[0]
+placeholder_token_id = tokenizer.convert_tokens_to_ids(args.placeholder_token)
+
+# Load the Stable Diffusion model
+# Load models and create wrapper for stable diffusion
+# pipeline = StableDiffusionPipeline.from_pretrained(pretrained_model_name_or_path)
+# del pipeline
+text_encoder = CLIPTextModel.from_pretrained(
+    pretrained_model_name_or_path, subfolder="text_encoder"
+)
+vae = AutoencoderKL.from_pretrained(
+    pretrained_model_name_or_path, subfolder="vae"
+)
+unet = UNet2DConditionModel.from_pretrained(
+    pretrained_model_name_or_path, subfolder="unet"
+)
+
+# We have added the placeholder_token in the tokenizer so we resize the token embeddings here
+# this will a new embedding vector in the token embeddings for our placeholder_token
+text_encoder.resize_token_embeddings(len(tokenizer))
+
+# Initialise the newly added placeholder token with the embeddings of the initializer token
+token_embeds = text_encoder.get_input_embeddings().weight.data
+token_embeds[placeholder_token_id] = token_embeds[initializer_token_id]
+
+# In Textual-Inversion we only train the newly added embedding vector
+#  so lets freeze rest of the model parameters here
+
+
+def freeze_params(params):
+    for param in params:
+        param.requires_grad = False
+
+
+# Freeze vae and unet
+freeze_params(vae.parameters())
+freeze_params(unet.parameters())
+# Freeze all parameters except for the token embeddings in text encoder
+params_to_freeze = itertools.chain(
+    text_encoder.text_model.encoder.parameters(),
+    text_encoder.text_model.final_layer_norm.parameters(),
+    text_encoder.text_model.embeddings.position_embedding.parameters(),
+)
+freeze_params(params_to_freeze)
+
+
+# Move vae and unet to device
+# For the dynamo path default compilation device is `cpu`, since torch-mlir
+# supports only that. Therefore, convert to device only for PyTorch path.
+if not args.use_torchdynamo:
+    vae.to(args.device)
+    unet.to(args.device)
+
+# Keep vae in eval mode as we don't train it
+vae.eval()
+# Keep unet in train mode to enable gradient checkpointing
+unet.train()
+
+
+class VaeModel(torch.nn.Module):
+    def __init__(self):
+        super().__init__()
+        self.vae = vae
+
+    def forward(self, input):
+        x = self.vae.encode(input, return_dict=False)[0]
+        return x
+
+
+class UnetModel(torch.nn.Module):
+    def __init__(self):
+        super().__init__()
+        self.unet = unet
+
+    def forward(self, x, y, z):
+        return self.unet.forward(x, y, z, return_dict=False)[0]
+
+
+shark_vae = VaeModel()
+shark_unet = UnetModel()
+
+####### Creating our training data ########
+
+# Let's create the Dataset and Dataloader
+train_dataset = TextualInversionDataset(
+    data_root=save_path,
+    tokenizer=tokenizer,
+    size=vae.sample_size,
+    placeholder_token=args.placeholder_token,
+    repeats=100,
+    learnable_property=args.what_to_teach,  # Option selected above between object and style
+    center_crop=False,
+    set="train",
+)
+
+
+def create_dataloader(train_batch_size=1):
+    return torch.utils.data.DataLoader(
+        train_dataset, batch_size=train_batch_size, shuffle=True
+    )
+
+
+# Create noise_scheduler for training
+noise_scheduler = DDPMScheduler.from_config(
+    pretrained_model_name_or_path, subfolder="scheduler"
+)
+
+######## Training ###########
+
+# Define hyperparameters for our training. If you are not happy with your results,
+# you can tune the `learning_rate` and the `max_train_steps`
+
+# Setting up all training args
+hyperparameters = {
+    "learning_rate": 5e-04,
+    "scale_lr": True,
+    "max_train_steps": args.training_steps,
+    "save_steps": args.save_steps,
+    "train_batch_size": args.train_batch_size,
+    "gradient_accumulation_steps": 1,
+    "gradient_checkpointing": True,
+    "mixed_precision": "fp16",
+    "seed": 42,
+    "output_dir": "sd-concept-output",
+}
+# creating output directory
+cwd = os.getcwd()
+out_dir = os.path.join(cwd, hyperparameters["output_dir"])
+while not os.path.exists(str(out_dir)):
+    try:
+        os.mkdir(out_dir)
+    except OSError as error:
+        print("Output directory not created")
+
+###### Torch-MLIR Compilation ######
+
+
+def _remove_nones(fx_g: torch.fx.GraphModule) -> List[int]:
+    removed_indexes = []
+    for node in fx_g.graph.nodes:
+        if node.op == "output":
+            assert (
+                len(node.args) == 1
+            ), "Output node must have a single argument"
+            node_arg = node.args[0]
+            if isinstance(node_arg, (list, tuple)):
+                node_arg = list(node_arg)
+                node_args_len = len(node_arg)
+                for i in range(node_args_len):
+                    curr_index = node_args_len - (i + 1)
+                    if node_arg[curr_index] is None:
+                        removed_indexes.append(curr_index)
+                        node_arg.pop(curr_index)
+                node.args = (tuple(node_arg),)
+                break
+
+    if len(removed_indexes) > 0:
+        fx_g.graph.lint()
+        fx_g.graph.eliminate_dead_code()
+        fx_g.recompile()
+    removed_indexes.sort()
+    return removed_indexes
+
+
+def _unwrap_single_tuple_return(fx_g: torch.fx.GraphModule) -> bool:
+    """
+    Replace tuple with tuple element in functions that return one-element tuples.
+    Returns true if an unwrapping took place, and false otherwise.
+    """
+    unwrapped_tuple = False
+    for node in fx_g.graph.nodes:
+        if node.op == "output":
+            assert (
+                len(node.args) == 1
+            ), "Output node must have a single argument"
+            node_arg = node.args[0]
+            if isinstance(node_arg, tuple):
+                if len(node_arg) == 1:
+                    node.args = (node_arg[0],)
+                    unwrapped_tuple = True
+                    break
+
+    if unwrapped_tuple:
+        fx_g.graph.lint()
+        fx_g.recompile()
+    return unwrapped_tuple
+
+
+def _returns_nothing(fx_g: torch.fx.GraphModule) -> bool:
+    for node in fx_g.graph.nodes:
+        if node.op == "output":
+            assert (
+                len(node.args) == 1
+            ), "Output node must have a single argument"
+            node_arg = node.args[0]
+            if isinstance(node_arg, tuple):
+                return len(node_arg) == 0
+    return False
+
+
+def transform_fx(fx_g):
+    for node in fx_g.graph.nodes:
+        if node.op == "call_function":
+            if node.target in [
+                torch.ops.aten.empty,
+            ]:
+                # aten.empty should be filled with zeros.
+                if node.target in [torch.ops.aten.empty]:
+                    with fx_g.graph.inserting_after(node):
+                        new_node = fx_g.graph.call_function(
+                            torch.ops.aten.zero_,
+                            args=(node,),
+                        )
+                        node.append(new_node)
+                        node.replace_all_uses_with(new_node)
+                        new_node.args = (node,)
+
+    fx_g.graph.lint()
+
+
+@make_simple_dynamo_backend
+def refbackend_torchdynamo_backend(
+    fx_graph: torch.fx.GraphModule, example_inputs: List[torch.Tensor]
+):
+    # handling usage of empty tensor without initializing
+    transform_fx(fx_graph)
+    fx_graph.recompile()
+    if _returns_nothing(fx_graph):
+        return fx_graph
+    removed_none_indexes = _remove_nones(fx_graph)
+    was_unwrapped = _unwrap_single_tuple_return(fx_graph)
+
+    mlir_module = torch_mlir.compile(
+        fx_graph, example_inputs, output_type="linalg-on-tensors"
+    )
+
+    bytecode_stream = BytesIO()
+    mlir_module.operation.write_bytecode(bytecode_stream)
+    bytecode = bytecode_stream.getvalue()
+
+    shark_module = SharkInference(
+        mlir_module=bytecode, device=args.device, mlir_dialect="tm_tensor"
+    )
+    shark_module.compile()
+
+    def compiled_callable(*inputs):
+        inputs = [x.numpy() for x in inputs]
+        result = shark_module("forward", inputs)
+        if was_unwrapped:
+            result = [
+                result,
+            ]
+        if not isinstance(result, list):
+            result = torch.from_numpy(result)
+        else:
+            result = tuple(torch.from_numpy(x) for x in result)
+            result = list(result)
+            for removed_index in removed_none_indexes:
+                result.insert(removed_index, None)
+            result = tuple(result)
+        return result
+
+    return compiled_callable
+
+
+def predictions(torch_func, jit_func, batchA, batchB):
+    res = jit_func(batchA.numpy(), batchB.numpy())
+    if res is not None:
+        prediction = res
+    else:
+        prediction = None
+    return prediction
+
+
+logger = logging.getLogger(__name__)
+
+
+# def save_progress(text_encoder, placeholder_token_id, accelerator, save_path):
+def save_progress(text_encoder, placeholder_token_id, save_path):
+    logger.info("Saving embeddings")
+    learned_embeds = (
+        # accelerator.unwrap_model(text_encoder)
+        text_encoder.get_input_embeddings().weight[placeholder_token_id]
+    )
+    learned_embeds_dict = {
+        args.placeholder_token: learned_embeds.detach().cpu()
+    }
+    torch.save(learned_embeds_dict, save_path)
+
+
+train_batch_size = hyperparameters["train_batch_size"]
+gradient_accumulation_steps = hyperparameters["gradient_accumulation_steps"]
+learning_rate = hyperparameters["learning_rate"]
+if hyperparameters["scale_lr"]:
+    learning_rate = (
+        learning_rate
+        * gradient_accumulation_steps
+        * train_batch_size
+        # * accelerator.num_processes
+    )
+
+# Initialize the optimizer
+optimizer = torch.optim.AdamW(
+    text_encoder.get_input_embeddings().parameters(),  # only optimize the embeddings
+    lr=learning_rate,
+)
+
+
+# Training function
+def train_func(batch_pixel_values, batch_input_ids):
+    # Convert images to latent space
+    latents = shark_vae(batch_pixel_values).sample().detach()
+    latents = latents * 0.18215
+
+    # Sample noise that we'll add to the latents
+    noise = torch.randn_like(latents)
+    bsz = latents.shape[0]
+    # Sample a random timestep for each image
+    timesteps = torch.randint(
+        0,
+        noise_scheduler.num_train_timesteps,
+        (bsz,),
+        device=latents.device,
+    ).long()
+
+    # Add noise to the latents according to the noise magnitude at each timestep
+    # (this is the forward diffusion process)
+    noisy_latents = noise_scheduler.add_noise(latents, noise, timesteps)
+
+    # Get the text embedding for conditioning
+    encoder_hidden_states = text_encoder(batch_input_ids)[0]
+
+    # Predict the noise residual
+    noise_pred = shark_unet(
+        noisy_latents,
+        timesteps,
+        encoder_hidden_states,
+    )
+
+    # Get the target for loss depending on the prediction type
+    if noise_scheduler.config.prediction_type == "epsilon":
+        target = noise
+    elif noise_scheduler.config.prediction_type == "v_prediction":
+        target = noise_scheduler.get_velocity(latents, noise, timesteps)
+    else:
+        raise ValueError(
+            f"Unknown prediction type {noise_scheduler.config.prediction_type}"
+        )
+
+    loss = (
+        F.mse_loss(noise_pred, target, reduction="none").mean([1, 2, 3]).mean()
+    )
+    loss.backward()
+
+    # Zero out the gradients for all token embeddings except the newly added
+    # embeddings for the concept, as we only want to optimize the concept embeddings
+    grads = text_encoder.get_input_embeddings().weight.grad
+    # Get the index for tokens that we want to zero the grads for
+    index_grads_to_zero = torch.arange(len(tokenizer)) != placeholder_token_id
+    grads.data[index_grads_to_zero, :] = grads.data[
+        index_grads_to_zero, :
+    ].fill_(0)
+
+    optimizer.step()
+    optimizer.zero_grad()
+
+    return loss
+
+
+def training_function():
+    max_train_steps = hyperparameters["max_train_steps"]
+    output_dir = hyperparameters["output_dir"]
+    gradient_checkpointing = hyperparameters["gradient_checkpointing"]
+
+    train_dataloader = create_dataloader(train_batch_size)
+
+    # We need to recalculate our total training steps as the size of the training dataloader may have changed.
+    num_update_steps_per_epoch = math.ceil(
+        len(train_dataloader) / gradient_accumulation_steps
+    )
+    num_train_epochs = math.ceil(max_train_steps / num_update_steps_per_epoch)
+
+    # Train!
+    total_batch_size = (
+        train_batch_size
+        * gradient_accumulation_steps
+        # train_batch_size * accelerator.num_processes * gradient_accumulation_steps
+    )
+
+    logger.info("***** Running training *****")
+    logger.info(f"  Num examples = {len(train_dataset)}")
+    logger.info(f"  Instantaneous batch size per device = {train_batch_size}")
+    logger.info(
+        f"  Total train batch size (w. parallel, distributed & accumulation) = {total_batch_size}"
+    )
+    logger.info(
+        f"  Gradient Accumulation steps = {gradient_accumulation_steps}"
+    )
+    logger.info(f"  Total optimization steps = {max_train_steps}")
+    # Only show the progress bar once on each machine.
+    progress_bar = tqdm(
+        # range(max_train_steps), disable=not accelerator.is_local_main_process
+        range(max_train_steps)
+    )
+    progress_bar.set_description("Steps")
+    global_step = 0
+
+    params_ = [i for i in text_encoder.get_input_embeddings().parameters()]
+    if args.use_torchdynamo:
+        print("******** TRAINING STARTED - TORCHYDNAMO PATH ********")
+    else:
+        print("******** TRAINING STARTED - PYTORCH PATH ********")
+    print("Initial weights:")
+    print(params_, params_[0].shape)
+
+    for epoch in range(num_train_epochs):
+        text_encoder.train()
+        for step, batch in enumerate(train_dataloader):
+            if args.use_torchdynamo:
+                dynamo_callable = dynamo.optimize(
+                    refbackend_torchdynamo_backend
+                )(train_func)
+                lam_func = lambda x, y: dynamo_callable(
+                    torch.from_numpy(x), torch.from_numpy(y)
+                )
+                loss = predictions(
+                    train_func,
+                    lam_func,
+                    batch["pixel_values"],
+                    batch["input_ids"],
+                    # params[0].detach(),
+                )
+            else:
+                loss = train_func(batch["pixel_values"], batch["input_ids"])
+            print(loss)
+
+            # Checks if the accelerator has performed an optimization step behind the scenes
+            progress_bar.update(1)
+            global_step += 1
+            if global_step % hyperparameters["save_steps"] == 0:
+                save_path = os.path.join(
+                    output_dir,
+                    f"learned_embeds-step-{global_step}.bin",
+                )
+                save_progress(
+                    text_encoder,
+                    placeholder_token_id,
+                    save_path,
+                )
+
+            logs = {"loss": loss.detach().item()}
+            progress_bar.set_postfix(**logs)
+
+            if global_step >= max_train_steps:
+                break
+
+    # Create the pipeline using using the trained modules and save it.
+    params__ = [i for i in text_encoder.get_input_embeddings().parameters()]
+    print("******** TRAINING PROCESS FINISHED ********")
+    print("Updated weights:")
+    print(params__, params__[0].shape)
+    pipeline = StableDiffusionPipeline.from_pretrained(
+        pretrained_model_name_or_path,
+        # text_encoder=accelerator.unwrap_model(text_encoder),
+        text_encoder=text_encoder,
+        tokenizer=tokenizer,
+        vae=vae,
+        unet=unet,
+    )
+    pipeline.save_pretrained(output_dir)
+    # Also save the newly trained embeddings
+    save_path = os.path.join(output_dir, f"learned_embeds.bin")
+    save_progress(text_encoder, placeholder_token_id, save_path)
+
+
+training_function()
+
+for param in itertools.chain(unet.parameters(), text_encoder.parameters()):
+    if param.grad is not None:
+        del param.grad  # free some memory
+    torch.cuda.empty_cache()
+
+# Set up the pipeline
+from diffusers import DPMSolverMultistepScheduler
+
+pipe = StableDiffusionPipeline.from_pretrained(
+    hyperparameters["output_dir"],
+    scheduler=DPMSolverMultistepScheduler.from_pretrained(
+        hyperparameters["output_dir"], subfolder="scheduler"
+    ),
+)
+if not args.use_torchdynamo:
+    pipe.to(args.device)
+
+# Run the Stable Diffusion pipeline
+# Don't forget to use the placeholder token in your prompt
+
+all_images = []
+for _ in range(args.num_inference_samples):
+    images = pipe(
+        [args.prompt],
+        num_inference_steps=args.inference_steps,
+        guidance_scale=7.5,
+    ).images
+    all_images.extend(images)
+
+output_path = os.path.abspath(os.path.join(os.getcwd(), args.output_dir))
+if not os.path.isdir(args.output_dir):
+    os.mkdir(args.output_dir)
+
+[
+    image.save(f"{args.output_dir}/{i}.jpeg")
+    for i, image in enumerate(all_images)
+]
--- a/shark/iree_utils/compile_utils.py
+++ b/shark/iree_utils/compile_utils.py
@@ -53,10 +53,10 @@ def get_iree_device_args(device, extra_args=[]):
 # Get the iree-compiler arguments given frontend.
 def get_iree_frontend_args(frontend):
    if frontend in ["torch", "pytorch", "linalg"]:
-        return ["--iree-llvm-target-cpu-features=host"]
+        return ["--iree-llvmcpu-target-cpu-features=host"]
    elif frontend in ["tensorflow", "tf", "mhlo"]:
        return [
-            "--iree-llvm-target-cpu-features=host",
+            "--iree-llvmcpu-target-cpu-features=host",
            "--iree-mhlo-demote-i64-to-i32=false",
            "--iree-flow-demote-i64-to-i32",
        ]
--- a/shark/iree_utils/cpu_utils.py
+++ b/shark/iree_utils/cpu_utils.py
@@ -44,4 +44,4 @@ def get_iree_cpu_args():
        error_message = f"OS Type f{os_name} not supported and triple can't be determined, open issue to dSHARK team please :)"
        raise Exception(error_message)
    print(f"Target triple found:{target_triple}")
-    return [f"-iree-llvm-target-triple={target_triple}"]
+    return [f"--iree-llvmcpu-target-triple={target_triple}"]
--- a/tank/all_models.csv
+++ b/tank/all_models.csv
@@ -5,7 +5,7 @@ bert-base-uncased,mhlo,tf,1e-2,1e-3,default,None,False,False,False,"","enabled_w
 camembert-base,mhlo,tf,1e-2,1e-3,default,None,True,True,True,"",""
 dbmdz/convbert-base-turkish-cased,mhlo,tf,1e-2,1e-3,default,nhcw-nhwc,True,True,False,"https://github.com/iree-org/iree/issues/9971",""
 distilbert-base-uncased,mhlo,tf,1e-2,1e-3,default,None,False,False,False,"",""
-facebook/convnext-tiny-224,mhlo,tf,1e-2,1e-3,tf_vit,nhcw-nhwc,True,True,False,"https://github.com/nod-ai/SHARK/issues/311 & https://github.com/nod-ai/SHARK/issues/342",""
+facebook/convnext-tiny-224,mhlo,tf,1e-2,1e-3,tf_vit,nhcw-nhwc,True,True,False,"https://github.com/nod-ai/SHARK/issues/311 & https://github.com/nod-ai/SHARK/issues/342","macos"
 funnel-transformer/small,mhlo,tf,1e-2,1e-3,default,None,True,True,False,"https://github.com/nod-ai/SHARK/issues/201",""
 google/electra-small-discriminator,mhlo,tf,1e-2,1e-3,default,None,False,False,False,"",""
 google/mobilebert-uncased,mhlo,tf,1e-2,1e-3,default,None,True,False,False,"Fails during iree-compile",""
@@ -18,8 +18,8 @@ alexnet,linalg,torch,1e-2,1e-3,default,None,True,True,False,"https://github.com/
 bert-base-cased,linalg,torch,1e-2,1e-3,default,None,False,False,False,"",""
 bert-base-uncased,linalg,torch,1e-2,1e-3,default,None,False,False,False,"",""
 bert-base-uncased_fp16,linalg,torch,1e-1,1e-1,default,None,True,False,True,"",""
-bert-large-uncased,linalg,torch,1e-2,1e-3,default,None,True,True,True,"disabled until generateable",""
-bert-large-uncased,mhlo,tf,1e-2,1e-3,default,None,True,True,True,"disabled until generatedable",""
+bert-large-uncased,linalg,torch,1e-2,1e-3,default,None,False,False,False,"",""
+bert-large-uncased,mhlo,tf,1e-2,1e-3,default,None,False,False,False,"",""
 facebook/deit-small-distilled-patch16-224,linalg,torch,1e-2,1e-3,default,nhcw-nhwc,False,True,False,"Fails during iree-compile.",""
 google/vit-base-patch16-224,linalg,torch,1e-2,1e-3,default,nhcw-nhwc,False,True,False,"https://github.com/nod-ai/SHARK/issues/311",""
 microsoft/beit-base-patch16-224-pt22k-ft22k,linalg,torch,1e-2,1e-3,default,nhcw-nhwc,False,True,False,"https://github.com/nod-ai/SHARK/issues/390",""
--- a/tank/examples/MiniLM_tf/huggingface_MiniLM_run.py
+++ b/tank/examples/MiniLM_tf/huggingface_MiniLM_run.py
@@ -63,7 +63,7 @@ if __name__ == "__main__":
    # Compile the model using IREE
    backend = "dylib-llvm-aot"
    args = [
-        "--iree-llvm-target-cpu-features=host",
+        "--iree-llvmcpu-target-cpu-features=host",
        "--iree-mhlo-demote-i64-to-i32=false",
        "--iree-flow-demote-i64-to-i32",
    ]
--- a/tank/examples/bert_fine_tuning/bert_fine_tune_tf.py
+++ b/tank/examples/bert_fine_tuning/bert_fine_tune_tf.py
@@ -136,7 +136,7 @@ if __name__ == "__main__":
    backend = "dylib-llvm-aot"
    if backend == "dylib-llvm-aot":
        args = [
-            "--iree-llvm-target-cpu-features=host",
+            "--iree-llvmcpu-target-cpu-features=host",
            "--iree-mhlo-demote-i64-to-i32=false",
            "--iree-flow-demote-i64-to-i32",
        ]
--- a/tank/examples/bert_tf/bert_large_run.py
+++ b/tank/examples/bert_tf/bert_large_run.py
@@ -83,7 +83,7 @@ if __name__ == "__main__":
    # Compile the model using IREE
    backend = "dylib-llvm-aot"
    args = [
-        "--iree-llvm-target-cpu-features=host",
+        "--iree-llvmcpu-target-cpu-features=host",
        "--iree-mhlo-demote-i64-to-i32=false",
        "--iree-stream-resource-index-bits=64",
        "--iree-vm-target-index-bits=64",
--- a/tank/examples/bert_tf/bert_small_run.py
+++ b/tank/examples/bert_tf/bert_small_run.py
@@ -79,7 +79,7 @@ if __name__ == "__main__":
    # Compile the model using IREE
    backend = "dylib-llvm-aot"
    args = [
-        "--iree-llvm-target-cpu-features=host",
+        "--iree-llvmcpu-target-cpu-features=host",
        "--iree-mhlo-demote-i64-to-i32=false",
        "--iree-flow-demote-i64-to-i32",
    ]
Author	SHA1	Message	Date
m68k-fr	16ad7d57a3	[WebUi] txt2img_ui: Import png metadata (#1147 )	2023-03-10 16:26:34 -08:00
Anush Elangovan	c561ebf43c	Drop the torch-mlir pin Seems to work now with top of master	2023-03-10 15:39:04 -08:00
Prashant Kumar	97fdff7f19	Add instructions how to run the LLaMA model. (#1168 ) * Add instructions how to run the LLaMA model. * Update README.md	2023-03-10 12:36:37 -08:00
Anush Elangovan	ce6d82eab2	Fix bloom lint	2023-03-10 11:53:08 -08:00
Abhishek Varma	b8f4b18951	[SD] Use dynamic stencil HF repo id -- This commit removes the hardcoded HF ID for Stencil and instead utilizes a dynamic instantiation of HF model. Signed-off-by: Abhishek Varma <abhishek@nod-labs.com>	2023-03-10 23:31:45 +05:30
Eliasj42	b23d3aa584	added more memory efficient method to run large bloom models with sharded blooms (#1165 ) Co-authored-by: Elias Joseph <elias@nod-labs.com>	2023-03-10 09:32:56 -08:00
Vivek Khandelwal	495670d9b6	Fix SD fine tuning script device arg usage	2023-03-10 18:37:53 +05:30
Boian Petkantchin	815e23a0b8	Update iree-compile flags --iree-llvm-xxx -> --iree-llvmcpu-xxx (#1164 )	2023-03-09 11:31:50 -08:00
Boian Petkantchin	783538fe11	Move linting opts from github workflow to config files This helps development where you can be sure that running locally black . flake8 . will do the same as in the github job.	2023-03-09 10:46:30 -08:00
Boian Petkantchin	996c645f6a	In SD don't include device path in vmfb filename Include only the driver name instead.	2023-03-09 10:45:32 -08:00
m68k-fr	1f7d249a62	Use utf-8 format for imgs_details.csv	2023-03-09 16:15:58 +05:30
jinchen62	7f6c9a2dc2	Add an inpainting option for only masked area (#1154 )	2023-03-07 09:46:05 -08:00
Eliasj42	93891984f3	made sharded bloom example more user friendly (#1153 ) Co-authored-by: Elias Joseph <elias@nod-labs.com>	2023-03-06 10:23:48 -08:00
Vivek Khandelwal	cc0ef54e0e	Fix Stable diffusion fine tuning script	2023-03-06 17:52:16 +05:30
Daniel Garvey	812152485d	temporarily xfail tiny convnext macos (#1142 )	2023-03-03 13:30:56 -06:00
Vivek Khandelwal	0816fb403a	Add Stable diffusion fine tuning script This commit adds the sd fine tuning script which runs through the torchdynamo path.	2023-03-03 21:59:00 +05:30
Gaurav Shukla	4f171772be	[SD] Fix SD web flags Signed-Off-by: Gaurav Shukla <gaurav@nod-labs.com>	2023-03-03 21:55:40 +05:30
mariecwhite	a52331d4aa	Install IREE pre-releases (#1139 )	2023-03-02 23:17:56 -06:00
yzhang93	ad821a1fc8	Use old torch-mlir package to avoid crash on rdna2 (#1137 )	2023-03-02 18:16:58 -08:00
Ean Garvey	116b128802	Use nightly shark_tank for test-models (#1133 ) * Use nightly shark_tank for test-models * Update all_models.csv	2023-03-02 12:33:36 -06:00