Add the int8 quantized model.

2026-04-25 03:00:12 -04:00 · 2023-03-21 10:37:34 +00:00
parent 420366c1b8
commit c2360303f0
5 changed files with 35 additions and 2 deletions
--- a/apps/stable_diffusion/scripts/txt2img.py
+++ b/apps/stable_diffusion/scripts/txt2img.py
@@ -205,6 +205,7 @@ if __name__ == "__main__":
        low_cpu_mem_usage=args.low_cpu_mem_usage,
        debug=args.import_debug if args.import_mlir else False,
        use_lora=args.use_lora,
+        use_quantize=args.use_quantize,
    )

    for current_batch in range(args.batch_count):
--- a/apps/stable_diffusion/src/models/model_wrappers.py
+++ b/apps/stable_diffusion/src/models/model_wrappers.py
@@ -100,7 +100,8 @@ class SharkifyStableDiffusionModel:
        is_inpaint: bool = False,
        is_upscaler: bool = False,
        use_stencil: str = None,
-        use_lora: str = ""
+        use_lora: str = "",
+        use_quantize: str = None,
    ):
        self.check_params(max_len, width, height)
        self.max_len = max_len
@@ -108,6 +109,7 @@ class SharkifyStableDiffusionModel:
        self.width = width // 8
        self.batch_size = batch_size
        self.custom_weights = custom_weights
+        self.use_quantize = use_quantize
        if custom_weights != "":
            assert custom_weights.lower().endswith(
                (".ckpt", ".safetensors")
@@ -570,7 +572,12 @@ class SharkifyStableDiffusionModel:
            compiled_controlnet = self.get_control_net()
            compiled_controlled_unet = self.get_controlled_unet()
        else:
-            compiled_unet = self.get_unet()
+            # TODO: Plug the experimental "int8" support at right place.
+            if self.use_quantize == "int8":
+                from apps.stable_diffusion.src.models.opt_params import get_unet
+                compiled_unet = get_unet()
+            else:
+                compiled_unet = self.get_unet()
        if self.custom_vae != "":
            print("Plugging in custom Vae")
        compiled_vae = self.get_vae()
--- a/apps/stable_diffusion/src/models/opt_params.py
+++ b/apps/stable_diffusion/src/models/opt_params.py
@@ -20,6 +20,15 @@ hf_model_variant_map = {
    "stabilityai/stable-diffusion-2-inpainting": ["stablediffusion", "inpaint_v2"],
 }

+# TODO: Add the quantized model as a part model_db.json.
+# This is currently in experimental phase.
+def get_quantize_model():
+    bucket_key = "gs://shark_tank/prashant_nod"
+    model_key = "unet_int8"
+    iree_flags = get_opt_flags("unet", precision="fp16")
+    if args.height != 512 and args.width != 512 and args.max_length != 77:
+        sys.exit("The int8 quantized model currently requires the height and width to be 512, and max_length to be 77")
+    return bucket_key, model_key, iree_flags

 def get_variant_version(hf_model_id):
    return hf_model_variant_map[hf_model_id]
@@ -41,6 +50,12 @@ def get_unet():
    variant, version = get_variant_version(args.hf_model_id)
    # Tuned model is present only for `fp16` precision.
    is_tuned = "tuned" if args.use_tuned else "untuned"
+
+    # TODO: Get the quantize model from model_db.json
+    if args.use_quantize == "int8":
+        bk, mk, flags = get_quantize_model()
+        return get_shark_model(bk, mk, flags)
+
    if "vulkan" not in args.device and args.use_tuned:
        bucket_key = f"{variant}/{is_tuned}/{args.device}"
        model_key = f"{variant}/{version}/unet/{args.precision}/length_{args.max_length}/{is_tuned}/{args.device}"
--- a/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_utils.py
+++ b/apps/stable_diffusion/src/pipelines/pipeline_shark_stable_diffusion_utils.py
@@ -321,6 +321,7 @@ class StableDiffusionPipeline:
        use_stencil: str = None,
        use_lora: str = "",
        ddpm_scheduler: DDPMScheduler = None,
+        use_quantize=None,
    ):
        is_inpaint = cls.__name__ in [
            "InpaintPipeline",
@@ -349,6 +350,7 @@ class StableDiffusionPipeline:
                is_upscaler=is_upscaler,
                use_stencil=use_stencil,
                use_lora=use_lora,
+                use_quantize=use_quantize,
            )
            if cls.__name__ in [
                "Image2ImagePipeline",
--- a/apps/stable_diffusion/src/utils/stable_args.py
+++ b/apps/stable_diffusion/src/utils/stable_args.py
@@ -340,6 +340,14 @@ p.add_argument(
    help="Use standalone LoRA weight using a HF ID or a checkpoint file (~3 MB)",
 )

+p.add_argument(
+    "--use_quantize",
+    type=str,
+    default="none",
+    help="""Runs the quantized version of stable diffusion model. This is currently in experimental phase.
+            Currently, only runs the stable-diffusion-2-1-base model in int8 quantization.""",
+)
+
 ##############################################################################
 ### IREE - Vulkan supported flags
 ##############################################################################