deepdanbooru interrogator

author: Greendayle <Greendayle> 2022-10-05 18:50:10 +0000
committer: Greendayle <Greendayle> 2022-10-05 18:55:26 +0000
commit: 59a2b9e5afc27d2fda72069ca0635070535d18fe (patch)
tree: f63afb4de3427072e81212131488f42be0fcbef4
parent: 1eb588cbf19924333b88beaa1ac0041904966640 (diff)
download: stable-diffusion-webui-gfx803-59a2b9e5afc27d2fda72069ca0635070535d18fe.tar.gz
stable-diffusion-webui-gfx803-59a2b9e5afc27d2fda72069ca0635070535d18fe.tar.bz2
stable-diffusion-webui-gfx803-59a2b9e5afc27d2fda72069ca0635070535d18fe.zip
6 files changed, 91 insertions, 6 deletions
diff --git a/models/deepbooru/Put your deepbooru release project folder here.txt b/models/deepbooru/Put your deepbooru release project folder here.txt
new file mode 100644
index 00000000..e69de29b
--- /dev/null
+++ b/models/deepbooru/Put your deepbooru release project folder here.txt
diff --git a/modules/deepbooru.py b/modules/deepbooru.py
new file mode 100644
index 00000000..958b1c3d
--- /dev/null
+++ b/modules/deepbooru.py
@@ -0,0 +1,60 @@
+import os.path
+from concurrent.futures import ProcessPoolExecutor
+
+import numpy as np
+import deepdanbooru as dd
+import tensorflow as tf
+
+
+def _load_tf_and_return_tags(pil_image, threshold):
+    this_folder = os.path.dirname(__file__)
+    model_path = os.path.join(this_folder, '..', 'models', 'deepbooru', 'deepdanbooru-v3-20211112-sgd-e28')
+    if not os.path.exists(model_path):
+        return "Download https://github.com/KichangKim/DeepDanbooru/releases/download/v3-20211112-sgd-e28/deepdanbooru-v3-20211112-sgd-e28.zip unpack and put into models/deepbooru"
+
+    tags = dd.project.load_tags_from_project(model_path)
+    model = dd.project.load_model_from_project(
+        model_path, compile_model=True
+    )
+
+    width = model.input_shape[2]
+    height = model.input_shape[1]
+    image = np.array(pil_image)
+    image = tf.image.resize(
+        image,
+        size=(height, width),
+        method=tf.image.ResizeMethod.AREA,
+        preserve_aspect_ratio=True,
+    )
+    image = image.numpy()  # EagerTensor to np.array
+    image = dd.image.transform_and_pad_image(image, width, height)
+    image = image / 255.0
+    image_shape = image.shape
+    image = image.reshape((1, image_shape[0], image_shape[1], image_shape[2]))
+
+    y = model.predict(image)[0]
+
+    result_dict = {}
+
+    for i, tag in enumerate(tags):
+        result_dict[tag] = y[i]
+
+
+
+    result_tags_out = []
+    result_tags_print = []
+    for tag in tags:
+        if result_dict[tag] >= threshold:
+            result_tags_out.append(tag)
+            result_tags_print.append(f'{result_dict[tag]} {tag}')
+
+    print('\n'.join(sorted(result_tags_print, reverse=True)))
+
+    return ', '.join(result_tags_out)
+
+
+def get_deepbooru_tags(pil_image, threshold=0.5):
+    with ProcessPoolExecutor() as executor:
+        f = executor.submit(_load_tf_and_return_tags, pil_image, threshold)
+        ret = f.result()  # will rethrow any exceptions
+    return ret
+\ No newline at end of file
diff --git a/modules/ui.py b/modules/ui.py
index 20dc8c37..ae98219a 100644
--- a/modules/ui.py
+++ b/modules/ui.py
@@ -23,6 +23,7 @@ import gradio.utils
 import gradio.routes
 
 from modules import sd_hijack
+from modules.deepbooru import get_deepbooru_tags
 from modules.paths import script_path
 from modules.shared import opts, cmd_opts
 import modules.shared as shared
@@ -312,6 +313,11 @@ def interrogate(image):
     return gr_show(True) if prompt is None else prompt
 
 
+def interrogate_deepbooru(image):
+    prompt = get_deepbooru_tags(image)
+    return gr_show(True) if prompt is None else prompt
+
+
 def create_seed_inputs():
     with gr.Row():
         with gr.Box():
@@ -439,15 +445,17 @@ def create_toprow(is_img2img):
                     outputs=[],
                 )
 
-            with gr.Row():
+            with gr.Row(scale=1):
                 if is_img2img:
-                    interrogate = gr.Button('Interrogate', elem_id="interrogate")
+                    interrogate = gr.Button('Interrogate\nCLIP', elem_id="interrogate")
+                    deepbooru = gr.Button('Interrogate\nDeepBooru', elem_id="deepbooru")
                 else:
                     interrogate = None
+                    deepbooru = None
                 prompt_style_apply = gr.Button('Apply style', elem_id="style_apply")
                 save_style = gr.Button('Create style', elem_id="style_create")
 
-    return prompt, roll, prompt_style, negative_prompt, prompt_style2, submit, interrogate, prompt_style_apply, save_style, paste, token_counter, token_button
+    return prompt, roll, prompt_style, negative_prompt, prompt_style2, submit, interrogate, deepbooru, prompt_style_apply, save_style, paste, token_counter, token_button
 
 
 def setup_progressbar(progressbar, preview, id_part, textinfo=None):
@@ -476,7 +484,7 @@ def create_ui(wrap_gradio_gpu_call):
     import modules.txt2img
 
     with gr.Blocks(analytics_enabled=False) as txt2img_interface:
-        txt2img_prompt, roll, txt2img_prompt_style, txt2img_negative_prompt, txt2img_prompt_style2, submit, _, txt2img_prompt_style_apply, txt2img_save_style, paste, token_counter, token_button = create_toprow(is_img2img=False)
+        txt2img_prompt, roll, txt2img_prompt_style, txt2img_negative_prompt, txt2img_prompt_style2, submit, _, _, txt2img_prompt_style_apply, txt2img_save_style, paste, token_counter, token_button = create_toprow(is_img2img=False)
         dummy_component = gr.Label(visible=False)
 
         with gr.Row(elem_id='txt2img_progress_row'):
@@ -628,7 +636,7 @@ def create_ui(wrap_gradio_gpu_call):
             token_button.click(fn=update_token_counter, inputs=[txt2img_prompt, steps], outputs=[token_counter])
 
     with gr.Blocks(analytics_enabled=False) as img2img_interface:
-        img2img_prompt, roll, img2img_prompt_style, img2img_negative_prompt, img2img_prompt_style2, submit, img2img_interrogate, img2img_prompt_style_apply, img2img_save_style, paste, token_counter, token_button = create_toprow(is_img2img=True)
+        img2img_prompt, roll, img2img_prompt_style, img2img_negative_prompt, img2img_prompt_style2, submit, img2img_interrogate, img2img_deepbooru, img2img_prompt_style_apply, img2img_save_style, paste, token_counter, token_button = create_toprow(is_img2img=True)
 
         with gr.Row(elem_id='img2img_progress_row'):
             with gr.Column(scale=1):
@@ -785,6 +793,12 @@ def create_ui(wrap_gradio_gpu_call):
                 outputs=[img2img_prompt],
             )
 
+            img2img_deepbooru.click(
+                fn=interrogate_deepbooru,
+                inputs=[init_img],
+                outputs=[img2img_prompt],
+            )
+
             save.click(
                 fn=wrap_gradio_call(save_files),
                 _js="(x, y, z) => [x, y, selected_gallery_index()]",
diff --git a/requirements.txt b/requirements.txt
index 631fe616..cab101f8 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -23,3 +23,6 @@ resize-right
 torchdiffeq
 kornia
 lark
+deepdanbooru
+tensorflow
+tensorflow-io
diff --git a/requirements_versions.txt b/requirements_versions.txt
index fdff2687..811953c6 100644
--- a/requirements_versions.txt
+++ b/requirements_versions.txt
@@ -22,3 +22,6 @@ resize-right==0.0.2
 torchdiffeq==0.2.3
 kornia==0.6.7
 lark==1.1.2
+git+https://github.com/KichangKim/DeepDanbooru.git@edf73df4cdaeea2cf00e9ac08bd8a9026b7a7b26#egg=deepdanbooru[tensorflow]
+tensorflow==2.10.0
+tensorflow-io==0.27.0
diff --git a/style.css b/style.css
index 39586bf1..2fd351f9 100644
--- a/style.css
+++ b/style.css
@@ -103,7 +103,12 @@
 
 #style_apply, #style_create, #interrogate{
     margin: 0.75em 0.25em 0.25em 0.25em;
-    min-width: 3em;
+    min-width: 5em;
+}
+
+#style_apply, #style_create, #deepbooru{
+    margin: 0.75em 0.25em 0.25em 0.25em;
+    min-width: 5em;
 }
 
 #style_pos_col, #style_neg_col{
author	Greendayle <Greendayle>	2022-10-05 18:50:10 +0000
committer	Greendayle <Greendayle>	2022-10-05 18:55:26 +0000
commit	59a2b9e5afc27d2fda72069ca0635070535d18fe (patch)
tree	f63afb4de3427072e81212131488f42be0fcbef4
parent	1eb588cbf19924333b88beaa1ac0041904966640 (diff)
download	stable-diffusion-webui-gfx803-59a2b9e5afc27d2fda72069ca0635070535d18fe.tar.gz stable-diffusion-webui-gfx803-59a2b9e5afc27d2fda72069ca0635070535d18fe.tar.bz2 stable-diffusion-webui-gfx803-59a2b9e5afc27d2fda72069ca0635070535d18fe.zip