Spaces:

SakanaAI
/

Evo-Nishikie

Running on Zero

App Files Files Community

Create evo_nishikie_v1.py, app.py, safety_checker.py and requirements.txt

by yuki-imajuku - opened Jul 12

base: refs/heads/main

←

from: refs/pr/1

Discussion Files changed

+48

-94

Files changed (10) hide show

.gitattributes +0 -4
README.md +4 -4
app.py +30 -56
evo_nishikie_v1.py +7 -10
requirements.txt +6 -7
safety_checker.py +1 -1
sample1.jpg +0 -3
sample2.jpg +0 -3
sample3.jpg +0 -3
sample4.jpg +0 -3

.gitattributes CHANGED Viewed

@@ -33,7 +33,3 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
-sample1.jpg filter=lfs diff=lfs merge=lfs -text
-sample2.jpg filter=lfs diff=lfs merge=lfs -text
-sample3.jpg filter=lfs diff=lfs merge=lfs -text
-sample4.jpg filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

README.md CHANGED Viewed

@@ -1,12 +1,12 @@
 ---
-title: Evo-Nishikie
-emoji: 🐠
 colorFrom: red
 colorTo: indigo
 sdk: gradio
-sdk_version: 4.38.1
 app_file: app.py
 pinned: false
 ---
-Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

 ---
+title: Evo Nishikie
+emoji: 🌍
 colorFrom: red
 colorTo: indigo
 sdk: gradio
+sdk_version: 4.37.2
 app_file: app.py
 pinned: false
 ---
+Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

app.py CHANGED Viewed

@@ -1,25 +1,20 @@
 import random
-from PIL import Image, ImageFilter
-from controlnet_aux import LineartDetector
-from diffusers import EulerDiscreteScheduler
 import gradio as gr
 import numpy as np
 import spaces
 import torch
-# torch._inductor.config.conv_1x1_as_mm = True
-# torch._inductor.config.coordinate_descent_tuning = True
-# torch._inductor.config.epilogue_fusion = False
-# torch._inductor.config.coordinate_descent_check_all_directions = True
 from evo_nishikie_v1 import load_evo_nishikie
 DESCRIPTION = """# 🐟 Evo-Nishikie
-🤗 [モデル一覧](https://huggingface.co/SakanaAI) | 📝 [ブログ](https://sakana.ai/evo-ukiyoe/) | 🐦 [Twitter](https://twitter.com/SakanaAILabs)
 [Evo-Nishikie](https://huggingface.co/SakanaAI/Evo-Nishikie-v1)は[Sakana AI](https://sakana.ai/)が教育目的で開発した浮世絵に特化した画像生成モデルです。
-入力した単色摺の浮世絵（墨摺絵等）を日本語プロンプトに沿って多色摺の浮世絵（錦絵）風に変換した画像を生成することができます。より詳しくは、上記のブログをご参照ください。
 """
 if not torch.cuda.is_available():
     DESCRIPTION += "\n<p>Running on CPU 🥶 This demo may not work on CPU.</p>"
@@ -52,22 +47,8 @@ if SAFETY_CHECKER:
         return images, has_nsfw_concepts
-pipe = load_evo_nishikie(device)
-pipe.scheduler = EulerDiscreteScheduler.from_config(
-    pipe.scheduler.config, use_karras_sigmas=True,
-)
-pipe.to(device=device, dtype=torch.float16)
-# pipe.unet.to(memory_format=torch.channels_last)
-# pipe.controlnet.to(memory_format=torch.channels_last)
-# pipe.vae.to(memory_format=torch.channels_last)
-# # Compile the UNet, ControlNet and VAE.
-# pipe.unet = torch.compile(pipe.unet, mode="max-autotune", fullgraph=True)
-# pipe.controlnet = torch.compile(pipe.controlnet, mode="max-autotune", fullgraph=True)
-# pipe.vae.decode = torch.compile(pipe.vae.decode, mode="max-autotune", fullgraph=True)
-lineart_detector = LineartDetector.from_pretrained("lllyasviel/Annotators")
-image_filter = ImageFilter.MedianFilter(size=3)
-BINARY_THRESHOLD = 40
 def randomize_seed_fn(seed: int, randomize_seed: bool) -> int:
@@ -79,26 +60,24 @@ def randomize_seed_fn(seed: int, randomize_seed: bool) -> int:
 @spaces.GPU
 @torch.inference_mode()
 def generate(
-    input_image: Image.Image,
     prompt: str,
     seed: int = 0,
     randomize_seed: bool = False,
     progress=gr.Progress(track_tqdm=True),
 ):
-    lineart_image = lineart_detector(input_image, coarse=False, image_resolution=1024)
-    lineart_image_filtered = lineart_image.filter(image_filter)
-    conditioning_image = lineart_image_filtered.point(lambda p: 255 if p > BINARY_THRESHOLD else 0).convert("L")
     seed = int(randomize_seed_fn(seed, randomize_seed))
     generator = torch.Generator().manual_seed(seed)
     images = pipe(
-        prompt=prompt + "最高品質の輻の浮世絵。超詳細。",
-        negative_prompt="暗い",
-        image=conditioning_image,
-        guidance_scale=7.0,
-        controlnet_conditioning_scale=0.8,
-        num_inference_steps=35,
         generator=generator,
         num_images_per_prompt=NUM_IMAGES_PER_PROMPT,
         output_type="pil",
@@ -113,37 +92,34 @@ def generate(
 examples = [
-    ["./sample1.jpg", "女性がやかんと鍋を持ち、小屋の前に立っています。背景には室内で会話する人々がいます。"],
-    ["./sample2.jpg", "着物を着た女性が、赤ん坊を抱え、もう一人の子どもが手押し車を引いています。背景には木があります。"],
-    ["./sample3.jpg", "女性が花柄の着物を着ており、他の人物たちが座りながら会話しています。背景には家の内部があります。"],
-    ["./sample4.jpg", "花柄や模様入りの着物を着た男女が室内で集まり、煎茶の準備をしています。背景に木材の装飾があります。"],
 ]
 css = """
-.gradio-container{max-width: 1380px !important}
 h1{text-align:center}
 """
 with gr.Blocks(css=css) as demo:
     gr.Markdown(DESCRIPTION)
-    with gr.Row():
-        with gr.Column():
-            input_image = gr.Image(image_mode="RGB", type="pil", show_label=False)
-            prompt = gr.Textbox(placeholder="日本語でプロンプトを入力してください。", show_label=False)
-            submit = gr.Button()
-            with gr.Accordion("詳細設定", open=False):
-                seed = gr.Slider(label="シード値", minimum=0, maximum=MAX_SEED, step=1, value=0)
-                randomize_seed = gr.Checkbox(label="ランダムにシード値を決定", value=True)
-        with gr.Column():
-            result = gr.Image(label="Evo-Nishikieからの生成結果", type="pil", show_label=False)
-    gr.Examples(examples=examples, inputs=[input_image, prompt], outputs=[result, seed], fn=generate)
     gr.on(
         triggers=[
             submit.click,
         ],
         fn=generate,
         inputs=[
-            input_image,
             prompt,
             seed,
             randomize_seed,
         ],
@@ -154,8 +130,6 @@ with gr.Blocks(css=css) as demo:
                 本モデルの使用は、利用者の自己責任で行われ、その性能や結果については何ら保証されません。
                 Sakana AIは、本モデルの使用によって生じた直接的または間接的な損失に対して、結果に関わらず、一切の責任を負いません。
                 利用者は、本モデルの使用に伴うリスクを十分に理解し、自身の判断で使用することが必要です。
-                アップロードされた画像は画像生成のみに使用され、サーバー上に保存されることはありません。
-                出典：サンプル画像はすべて[日本古典籍データセット（国文学研究資料館蔵）『絵本玉かつら』](http://codh.rois.ac.jp/pmjt/book/200013861/)から引用しました。""")
 demo.queue().launch()

 import random
+from PIL import Image
+from controlnet_aux import CannyDetector
 import gradio as gr
 import numpy as np
 import spaces
 import torch
 from evo_nishikie_v1 import load_evo_nishikie
 DESCRIPTION = """# 🐟 Evo-Nishikie
+🤗 [モデル一覧](https://huggingface.co/SakanaAI) | 📚 [技術レポート](https://arxiv.org/abs/2403.13187) | 📝 [ブログ](https://sakana.ai/evosdxl-jp/) | 🐦 [Twitter](https://twitter.com/SakanaAILabs)
 [Evo-Nishikie](https://huggingface.co/SakanaAI/Evo-Nishikie-v1)は[Sakana AI](https://sakana.ai/)が教育目的で開発した浮世絵に特化した画像生成モデルです。
+入力した画像を日本語プロンプトに沿って浮世絵風に変換した画像を生成することができます。より詳しくは、上記のブログをご参照ください。
 """
 if not torch.cuda.is_available():
     DESCRIPTION += "\n<p>Running on CPU 🥶 This demo may not work on CPU.</p>"
         return images, has_nsfw_concepts
+pipe = load_evo_nishikie("cpu").to(device)
+canny_detector = CannyDetector()
 def randomize_seed_fn(seed: int, randomize_seed: bool) -> int:
 @spaces.GPU
 @torch.inference_mode()
 def generate(
     prompt: str,
+    input_image: Image.Image,
     seed: int = 0,
     randomize_seed: bool = False,
     progress=gr.Progress(track_tqdm=True),
 ):
+    pipe.to(device)
+    canny_image = canny_detector(input_image, image_resolution=1024)
     seed = int(randomize_seed_fn(seed, randomize_seed))
     generator = torch.Generator().manual_seed(seed)
     images = pipe(
+        prompt=prompt + "最高品質の輻の浮世絵。",
+        negative_prompt="暗い。",
+        image=canny_image,
+        guidance_scale=8.0,
+        controlnet_conditioning_scale=0.6,
+        num_inference_steps=50,
         generator=generator,
         num_images_per_prompt=NUM_IMAGES_PER_PROMPT,
         output_type="pil",
 examples = [
+    ["銀杏が色づく。草木が生えた地面と青空の富士山。", "https://sakana.ai/assets/nedo-grant/nedo_grant.jpeg"],
 ]
 css = """
+.gradio-container{max-width: 690px !important}
 h1{text-align:center}
 """
 with gr.Blocks(css=css) as demo:
     gr.Markdown(DESCRIPTION)
+    with gr.Group():
+        with gr.Row():
+            with gr.Column(scale=8.0):
+                prompt = gr.Textbox(placeholder="日本語でプロンプトを入力してください。", show_label=False)
+                input_image = gr.Image(image_mode="RGB", type="pil", show_label=False)
+            submit = gr.Button(scale=0)
+        result = gr.Image(label="Evo-Nishikieからの生成結果", type="pil", show_label=False)
+    with gr.Accordion("詳細設定", open=False):
+        seed = gr.Slider(label="シード値", minimum=0, maximum=MAX_SEED, step=1, value=0)
+        randomize_seed = gr.Checkbox(label="ランダムにシード値を決定", value=True)
+    gr.Examples(examples=examples, inputs=[prompt, input_image], outputs=[result, seed], fn=generate)
     gr.on(
         triggers=[
             submit.click,
         ],
         fn=generate,
         inputs=[
             prompt,
+            input_image,
             seed,
             randomize_seed,
         ],
                 本モデルの使用は、利用者の自己責任で行われ、その性能や結果については何ら保証されません。
                 Sakana AIは、本モデルの使用によって生じた直接的または間接的な損失に対して、結果に関わらず、一切の責任を負いません。
                 利用者は、本モデルの使用に伴うリスクを十分に理解し、自身の判断で使用することが必要です。
+                アップロードされた画像は画像生成のみに使用され、サーバー上に保存されることはありません。""")
 demo.queue().launch()

evo_nishikie_v1.py CHANGED Viewed

@@ -1,20 +1,23 @@
 import gc
 import os
 from typing import Dict, List, Union
 from diffusers import (
     ControlNetModel,
     StableDiffusionXLControlNetPipeline,
     UNet2DConditionModel,
 )
 from huggingface_hub import hf_hub_download
 import safetensors
 import torch
 from tqdm import tqdm
 from transformers import AutoTokenizer, CLIPTextModelWithProjection
-# Base models
 SDXL_REPO = "stabilityai/stable-diffusion-xl-base-1.0"
 DPO_REPO = "mhdang/dpo-sdxl-text2image-v1"
 JN_REPO = "RunDiffusion/Juggernaut-XL-v9"
@@ -121,7 +124,6 @@ def load_evo_nishikie(device="cuda") -> StableDiffusionXLControlNetPipeline:
     )
     jn_weights = split_conv_attn(load_from_pretrained(JN_REPO, device=device))
     jsdxl_weights = split_conv_attn(load_from_pretrained(JSDXL_REPO, device=device))
     # Merge base models
     tensors = [sdxl_weights, dpo_weights, jn_weights, jsdxl_weights]
     new_conv = merge_models(
@@ -142,14 +144,11 @@ def load_evo_nishikie(device="cuda") -> StableDiffusionXLControlNetPipeline:
             0.2198623756106564,
         ],
     )
-    # Delete no longer needed variables to free
     del sdxl_weights, dpo_weights, jn_weights, jsdxl_weights
     gc.collect()
     if "cuda" in device:
         torch.cuda.empty_cache()
-    # Instantiate UNet
     unet_config = UNet2DConditionModel.load_config(SDXL_REPO, subfolder="unet")
     unet = UNet2DConditionModel.from_config(unet_config).to(device=device)
     unet.load_state_dict({**new_conv, **new_attn})
@@ -177,11 +176,9 @@ def load_evo_nishikie(device="cuda") -> StableDiffusionXLControlNetPipeline:
         torch_dtype=torch.float16,
         variant="fp16",
     )
     # Load Evo-Ukiyoe weights
     pipe.load_lora_weights(UKIYOE_REPO)
     pipe.fuse_lora(lora_scale=1.0)
-    pipe = pipe.to(device, dtype=torch.float16)
-    return pipe

 import gc
+from io import BytesIO
 import os
 from typing import Dict, List, Union
+from PIL import Image
+from controlnet_aux import CannyDetector
 from diffusers import (
     ControlNetModel,
     StableDiffusionXLControlNetPipeline,
     UNet2DConditionModel,
 )
 from huggingface_hub import hf_hub_download
+import requests
 import safetensors
 import torch
 from tqdm import tqdm
 from transformers import AutoTokenizer, CLIPTextModelWithProjection
+# Base models (fine-tuned from SDXL-1.0)
 SDXL_REPO = "stabilityai/stable-diffusion-xl-base-1.0"
 DPO_REPO = "mhdang/dpo-sdxl-text2image-v1"
 JN_REPO = "RunDiffusion/Juggernaut-XL-v9"
     )
     jn_weights = split_conv_attn(load_from_pretrained(JN_REPO, device=device))
     jsdxl_weights = split_conv_attn(load_from_pretrained(JSDXL_REPO, device=device))
     # Merge base models
     tensors = [sdxl_weights, dpo_weights, jn_weights, jsdxl_weights]
     new_conv = merge_models(
             0.2198623756106564,
         ],
     )
     del sdxl_weights, dpo_weights, jn_weights, jsdxl_weights
     gc.collect()
     if "cuda" in device:
         torch.cuda.empty_cache()
     unet_config = UNet2DConditionModel.load_config(SDXL_REPO, subfolder="unet")
     unet = UNet2DConditionModel.from_config(unet_config).to(device=device)
     unet.load_state_dict({**new_conv, **new_attn})
         torch_dtype=torch.float16,
         variant="fp16",
     )
+    pipe = pipe.to(device, dtype=torch.float16)
     # Load Evo-Ukiyoe weights
     pipe.load_lora_weights(UKIYOE_REPO)
     pipe.fuse_lora(lora_scale=1.0)
+    return pipe

requirements.txt CHANGED Viewed

@@ -1,10 +1,9 @@
-torch
-torchvision
-accelerate==0.32.0
-controlnet-aux==0.0.9
-diffusers==0.29.2
-gradio==4.38.1
 sentencepiece==0.2.0
 transformers==4.42.3
-peft==0.11.1

+--extra-index-url https://download.pytorch.org/whl/cu121
+torch==2.3.1+cu121
+torchvision==0.18.1+cu121
+accelerate==0.31.0
+diffusers==0.26.0
+gradio==4.37.2
 sentencepiece==0.2.0
 transformers==4.42.3

safety_checker.py CHANGED Viewed

@@ -134,4 +134,4 @@ class StableDiffusionSafetyChecker(PreTrainedModel):
         images[has_nsfw_concepts] = 0.0  # black image
-        return images, has_nsfw_concepts


134
135	images[has_nsfw_concepts] = 0.0 # black image
136
137	+ return images, has_nsfw_concepts

sample1.jpg DELETED Viewed

Git LFS Details

SHA256: b9fe5a98203730cb506e915c42a1031b3cebe28f5fa9b5489d635a494ee26be8
Pointer size: 132 Bytes
Size of remote file: 2.8 MB

sample2.jpg DELETED Viewed

Git LFS Details

SHA256: 52803116a60fd53f8ed5f621f4296c66a3c761dce9589b6cdfb6acff6b269ab2
Pointer size: 132 Bytes
Size of remote file: 1.33 MB

sample3.jpg DELETED Viewed

Git LFS Details

SHA256: 7f7fc9aad3df3400a774527faf821b1e815f5a47fc1e53f0ed59a5a4aa305bd7
Pointer size: 132 Bytes
Size of remote file: 2.48 MB

sample4.jpg DELETED Viewed

Git LFS Details

SHA256: c27553237814625692eb344dc28d0e2cceb1b9176a95ab96a2c5ba64313ac162
Pointer size: 132 Bytes
Size of remote file: 2.67 MB