Real-Time-Latent-Consistency-Model

Runtime error

App Files Files Community

radames commited on Jun 20, 2024

Commit

947c4f5

1 Parent(s): e13bdf0

safety checker

Browse files

Files changed (2) hide show

server/pipelines/controlnetLoraSDXL-Lightning.py +14 -16
server/pipelines/utils/safety_checker.py +163 -0

server/pipelines/controlnetLoraSDXL-Lightning.py CHANGED Viewed

@@ -9,6 +9,7 @@ from diffusers import (
 from compel import Compel, ReturnedEmbeddingsType
 import torch
 from pipelines.utils.canny_gpu import SobelOperator
 from huggingface_hub import hf_hub_download
 from safetensors.torch import load_file
@@ -17,7 +18,6 @@ try:
 except:
     pass
-import psutil
 from config import Args
 from pydantic import BaseModel, Field
 from PIL import Image
@@ -35,7 +35,7 @@ default_prompt = "Portrait of The Terminator with , glare pose, detailed, intric
 default_negative_prompt = "blurry, low quality, render, 3D, oversaturated"
 page_content = """
 <h1 class="text-3xl font-bold">Real-Time Latent Consistency Model SDXL</h1>
-<h3 class="text-xl font-bold">SDXL-Lightining + LCM + LoRA + Controlnet</h3>
 <p class="text-sm">
     This demo showcases
     <a
@@ -84,9 +84,6 @@ class Pipeline:
         seed: int = Field(
             2159232, min=0, title="Seed", field="seed", hide=True, id="seed"
         )
-        steps: int = Field(
-            1, min=1, max=10, title="Steps", field="range", hide=True, id="steps"
-        )
         width: int = Field(
             1024, min=2, max=15, title="Width", disabled=True, hide=True, id="width"
         )
@@ -172,6 +169,8 @@ class Pipeline:
         )
     def __init__(self, args: Args, device: torch.device, torch_dtype: torch.dtype):
         if args.taesd:
             vae = AutoencoderTiny.from_pretrained(
@@ -204,7 +203,7 @@ class Pipeline:
         self.canny_torch = SobelOperator(device=device)
         self.pipe.set_progress_bar_config(disable=True)
-        self.pipe.to(device=device, dtype=torch_dtype).to(device)
         if args.sfast:
             from sfast.compilers.stable_diffusion_pipeline_compiler import (
@@ -242,7 +241,7 @@ class Pipeline:
                 control_image=[Image.new("RGB", (768, 768))],
             )
-    def predict(self, params: "Pipeline.InputParams") -> Image.Image:
         generator = torch.manual_seed(params.seed)
         prompt = params.prompt
@@ -265,7 +264,7 @@ class Pipeline:
         control_image = self.canny_torch(
             params.image, params.canny_low_threshold, params.canny_high_threshold
         )
-        steps = params.steps
         strength = params.strength
         if int(steps * strength) < 1:
             steps = math.ceil(1 / max(0.10, strength))
@@ -281,7 +280,7 @@ class Pipeline:
             negative_pooled_prompt_embeds=negative_pooled_prompt_embeds,
             generator=generator,
             strength=strength,
-            num_inference_steps=steps,
             guidance_scale=params.guidance_scale,
             width=params.width,
             height=params.height,
@@ -290,14 +289,13 @@ class Pipeline:
             control_guidance_start=params.controlnet_start,
             control_guidance_end=params.controlnet_end,
         )
-        nsfw_content_detected = (
-            results.nsfw_content_detected[0]
-            if "nsfw_content_detected" in results
-            else False
-        )
-        if nsfw_content_detected:
             return None
         result_image = results.images[0]
         if params.debug_canny:
             # paste control_image on top of result_image

 from compel import Compel, ReturnedEmbeddingsType
 import torch
 from pipelines.utils.canny_gpu import SobelOperator
+from pipelines.utils.safety_checker import SafetyChecker
 from huggingface_hub import hf_hub_download
 from safetensors.torch import load_file
 except:
     pass
 from config import Args
 from pydantic import BaseModel, Field
 from PIL import Image
 default_negative_prompt = "blurry, low quality, render, 3D, oversaturated"
 page_content = """
 <h1 class="text-3xl font-bold">Real-Time Latent Consistency Model SDXL</h1>
+<h3 class="text-xl font-bold">SDXL-Lightining + Controlnet</h3>
 <p class="text-sm">
     This demo showcases
     <a
         seed: int = Field(
             2159232, min=0, title="Seed", field="seed", hide=True, id="seed"
         )
         width: int = Field(
             1024, min=2, max=15, title="Width", disabled=True, hide=True, id="width"
         )
         )
     def __init__(self, args: Args, device: torch.device, torch_dtype: torch.dtype):
+        if args.safety_checker:
+            self.safety_checker = SafetyChecker(device=device.type)
         if args.taesd:
             vae = AutoencoderTiny.from_pretrained(
         self.canny_torch = SobelOperator(device=device)
         self.pipe.set_progress_bar_config(disable=True)
+        self.pipe.to(device=device, dtype=torch_dtype)
         if args.sfast:
             from sfast.compilers.stable_diffusion_pipeline_compiler import (
                 control_image=[Image.new("RGB", (768, 768))],
             )
+    def predict(self, params: "Pipeline.InputParams") -> Image.Image | None:
         generator = torch.manual_seed(params.seed)
         prompt = params.prompt
         control_image = self.canny_torch(
             params.image, params.canny_low_threshold, params.canny_high_threshold
         )
+        steps = NUM_STEPS
         strength = params.strength
         if int(steps * strength) < 1:
             steps = math.ceil(1 / max(0.10, strength))
             negative_pooled_prompt_embeds=negative_pooled_prompt_embeds,
             generator=generator,
             strength=strength,
+            num_inference_steps=NUM_STEPS,
             guidance_scale=params.guidance_scale,
             width=params.width,
             height=params.height,
             control_guidance_start=params.controlnet_start,
             control_guidance_end=params.controlnet_end,
         )
+        images = results.images
+        if self.safety_checker:
+            images, has_nsfw_concepts = self.safety_checker(images)
+        print(has_nsfw_concepts)
+        if any(has_nsfw_concepts):
             return None
         result_image = results.images[0]
         if params.debug_canny:
             # paste control_image on top of result_image

server/pipelines/utils/safety_checker.py ADDED Viewed

	@@ -0,0 +1,163 @@

+# Copyright 2023 The HuggingFace Team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import torch
+import torch.nn as nn
+from transformers import CLIPConfig, CLIPVisionModel, PreTrainedModel
+from PIL import Image
+def cosine_distance(image_embeds, text_embeds):
+    normalized_image_embeds = nn.functional.normalize(image_embeds)
+    normalized_text_embeds = nn.functional.normalize(text_embeds)
+    return torch.mm(normalized_image_embeds, normalized_text_embeds.t())
+class StableDiffusionSafetyChecker(PreTrainedModel):
+    config_class = CLIPConfig
+    _no_split_modules = ["CLIPEncoderLayer"]
+    def __init__(self, config: CLIPConfig):
+        super().__init__(config)
+        self.vision_model = CLIPVisionModel(config.vision_config)
+        self.visual_projection = nn.Linear(
+            config.vision_config.hidden_size, config.projection_dim, bias=False
+        )
+        self.concept_embeds = nn.Parameter(
+            torch.ones(17, config.projection_dim), requires_grad=False
+        )
+        self.special_care_embeds = nn.Parameter(
+            torch.ones(3, config.projection_dim), requires_grad=False
+        )
+        self.concept_embeds_weights = nn.Parameter(torch.ones(17), requires_grad=False)
+        self.special_care_embeds_weights = nn.Parameter(
+            torch.ones(3), requires_grad=False
+        )
+    @torch.no_grad()
+    def forward(self, clip_input, images):
+        pooled_output = self.vision_model(clip_input)[1]  # pooled_output
+        image_embeds = self.visual_projection(pooled_output)
+        # we always cast to float32 as this does not cause significant overhead and is compatible with bfloat16
+        special_cos_dist = (
+            cosine_distance(image_embeds, self.special_care_embeds)
+            .cpu()
+            .float()
+            .numpy()
+        )
+        cos_dist = (
+            cosine_distance(image_embeds, self.concept_embeds).cpu().float().numpy()
+        )
+        result = []
+        batch_size = image_embeds.shape[0]
+        for i in range(batch_size):
+            result_img = {
+                "special_scores": {},
+                "special_care": [],
+                "concept_scores": {},
+                "bad_concepts": [],
+            }
+            # increase this value to create a stronger `nfsw` filter
+            # at the cost of increasing the possibility of filtering benign images
+            adjustment = 0.0
+            for concept_idx in range(len(special_cos_dist[0])):
+                concept_cos = special_cos_dist[i][concept_idx]
+                concept_threshold = self.special_care_embeds_weights[concept_idx].item()
+                result_img["special_scores"][concept_idx] = round(
+                    concept_cos - concept_threshold + adjustment, 3
+                )
+                if result_img["special_scores"][concept_idx] > 0:
+                    result_img["special_care"].append(
+                        {concept_idx, result_img["special_scores"][concept_idx]}
+                    )
+                    adjustment = 0.01
+            for concept_idx in range(len(cos_dist[0])):
+                concept_cos = cos_dist[i][concept_idx]
+                concept_threshold = self.concept_embeds_weights[concept_idx].item()
+                result_img["concept_scores"][concept_idx] = round(
+                    concept_cos - concept_threshold + adjustment, 3
+                )
+                if result_img["concept_scores"][concept_idx] > 0:
+                    result_img["bad_concepts"].append(concept_idx)
+            result.append(result_img)
+        has_nsfw_concepts = [len(res["bad_concepts"]) > 0 for res in result]
+        return has_nsfw_concepts
+    @torch.no_grad()
+    def forward_onnx(self, clip_input: torch.FloatTensor, images: torch.FloatTensor):
+        pooled_output = self.vision_model(clip_input)[1]  # pooled_output
+        image_embeds = self.visual_projection(pooled_output)
+        special_cos_dist = cosine_distance(image_embeds, self.special_care_embeds)
+        cos_dist = cosine_distance(image_embeds, self.concept_embeds)
+        # increase this value to create a stronger `nsfw` filter
+        # at the cost of increasing the possibility of filtering benign images
+        adjustment = 0.0
+        special_scores = (
+            special_cos_dist - self.special_care_embeds_weights + adjustment
+        )
+        # special_scores = special_scores.round(decimals=3)
+        special_care = torch.any(special_scores > 0, dim=1)
+        special_adjustment = special_care * 0.01
+        special_adjustment = special_adjustment.unsqueeze(1).expand(
+            -1, cos_dist.shape[1]
+        )
+        concept_scores = (cos_dist - self.concept_embeds_weights) + special_adjustment
+        # concept_scores = concept_scores.round(decimals=3)
+        has_nsfw_concepts = torch.any(concept_scores > 0, dim=1)
+        images[has_nsfw_concepts] = 0.0  # black image
+        return images, has_nsfw_concepts
+class SafetyChecker:
+    def __init__(self, device="cuda"):
+        from transformers import CLIPFeatureExtractor
+        self.device = device
+        self.safety_checker = StableDiffusionSafetyChecker.from_pretrained(
+            "CompVis/stable-diffusion-safety-checker"
+        ).to(device)
+        self.feature_extractor = CLIPFeatureExtractor.from_pretrained(
+            "openai/clip-vit-base-patch32"
+        )
+    def __call__(
+        self, images: list[Image.Image]
+    ) -> tuple[list[Image.Image], list[bool]]:
+        safety_checker_input = self.feature_extractor(images, return_tensors="pt").to(
+            self.device
+        )
+        has_nsfw_concepts = self.safety_checker(
+            images=[images],
+            clip_input=safety_checker_input.pixel_values.to(self.device),
+        )
+        return images, has_nsfw_concepts