From 54acb47f418cb57cb21b16f4f48fd88836afbcb6 Mon Sep 17 00:00:00 2001 From: Vladimir Mandic Date: Tue, 26 Dec 2023 13:19:38 -0500 Subject: [PATCH] add faceid module --- CHANGELOG.md | 6 +++ installer.py | 3 +- scripts/faceid.py | 109 ++++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 117 insertions(+), 1 deletion(-) create mode 100644 scripts/faceid.py diff --git a/CHANGELOG.md b/CHANGELOG.md index a792bff42..9b8cd419d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -18,6 +18,12 @@ - **IP Adapter** - add support for `ip-adapter-plus_sd15`, `ip-adapter-plus-face_sd15` and `ip-adapter-full-face_sd15` - can now be used in *xyz-grid* + - **FaceID** + - also based on IP adapters, but with additional face detection and external embeddings calculation + - calculates face embeds based on input image and uses it to guide generation + - simply select from *scripts -> faceid* + - *experimental module*: requirements must be installed manually: + > pip install insightface ip_adapter - **Text-to-Video** - in text tab, select `text-to-video` script - supported models: ModelScope v1.7b, ZeroScope v1, ZeroScope v1.1, ZeroScope v2, ZeroScope v2 Dark, Potat v1 diff --git a/installer.py b/installer.py index 2a15bb4c4..432fda9be 100644 --- a/installer.py +++ b/installer.py @@ -363,7 +363,8 @@ def check_torch(): log.debug(f'Torch allowed: cuda={allow_cuda} rocm={allow_rocm} ipex={allow_ipex} diml={allow_directml} openvino={allow_openvino}') torch_command = os.environ.get('TORCH_COMMAND', '') xformers_package = os.environ.get('XFORMERS_PACKAGE', 'none') - install('onnxruntime onnxruntimegpu', 'onnxruntime', ignore=True) + if not installed('onnxruntime', quiet=True) and not installed('onnxruntime-gpu', quiet=True): # allow either + install('onnxruntime', 'onnxruntime', ignore=True) if torch_command != '': pass elif allow_cuda and (shutil.which('nvidia-smi') is not None or args.use_xformers or os.path.exists(os.path.join(os.environ.get('SystemRoot') or r'C:\Windows', 'System32', 'nvidia-smi.exe'))): diff --git a/scripts/faceid.py b/scripts/faceid.py new file mode 100644 index 000000000..4135ea952 --- /dev/null +++ b/scripts/faceid.py @@ -0,0 +1,109 @@ +import os +import cv2 +import torch +import numpy as np +import gradio as gr +import diffusers +import huggingface_hub as hf +from modules import scripts, processing, shared, devices + + +app = None +try: + import onnxruntime + from insightface.app import FaceAnalysis + from ip_adapter.ip_adapter_faceid import IPAdapterFaceID + ok = True +except Exception as e: + shared.log.error(f'FaceID: {e}') + ok = False + + +class Script(scripts.Script): + def title(self): + return 'FaceID' + + def show(self, is_img2img): + return ok if shared.backend == shared.Backend.DIFFUSERS else False + + # return signature is array of gradio components + def ui(self, _is_img2img): + with gr.Row(): + scale = gr.Slider(label='Scale', minimum=0.0, maximum=1.0, step=0.01, value=1.0) + with gr.Row(): + image = gr.Image(image_mode='RGB', label='Image', source='upload', type='pil', width=512) + return [scale, image] + + def run(self, p: processing.StableDiffusionProcessing, scale, image): # pylint: disable=arguments-differ, unused-argument + if not ok: + shared.log.error('FaceID: missing dependencies') + return None + if image is None: + shared.log.error('FaceID: no init_images') + return None + if shared.sd_model_type != 'sd': + shared.log.error('FaceID: base model not supported') + return None + + global app # pylint: disable=global-statement + if app is None: + shared.log.debug(f"ONNX: device={onnxruntime.get_device()} providers={onnxruntime.get_available_providers()}") + app = FaceAnalysis(name="buffalo_l", providers=['CUDAExecutionProvider', 'CPUExecutionProvider']) + onnxruntime.set_default_logger_severity(3) + app.prepare(ctx_id=0, det_thresh=0.5, det_size=(640, 640)) + + image = cv2.cvtColor(np.array(image), cv2.COLOR_RGB2BGR) + faces = app.get(image) + if len(faces) == 0: + shared.log.error('FaceID: no faces found') + return None + for face in faces: + shared.log.debug(f'FaceID face: score={face.det_score:.2f} gender={"female" if face.gender==0 else "male"} age={face.age} bbox={face.bbox}') + embeds = torch.from_numpy(faces[0].normed_embedding).unsqueeze(0) + + ip_ckpt = "h94/IP-Adapter-FaceID/ip-adapter-faceid_sd15.bin" + shared.log.debug(f'FaceID model load: {ip_ckpt}') + folder, filename = os.path.split(ip_ckpt) + model_path = hf.hf_hub_download(repo_id=folder, filename=filename, cache_dir=shared.opts.diffusers_dir) + if model_path is None: + shared.log.error(f'FaceID: model download failed: {ip_ckpt}') + return None + + processing.process_init(p) + shared.sd_model.scheduler = diffusers.DDIMScheduler( + num_train_timesteps=1000, + beta_start=0.00085, + beta_end=0.012, + beta_schedule="scaled_linear", + clip_sample=False, + set_alpha_to_one=False, + steps_offset=1, + ) + ip_model = IPAdapterFaceID(shared.sd_model, model_path, devices.device) + ip_model_dict = { + 'prompt': p.all_prompts[0], + 'negative_prompt': p.all_negative_prompts[0], + 'num_samples': p.batch_size, + 'width': p.width, + 'height': p.height, + 'num_inference_steps': p.steps, + 'scale': scale, + 'guidance_scale': p.cfg_scale, + 'seed': int(p.all_seeds[0]), + 'faceid_embeds': None, + } + shared.log.debug(f'FaceID args: {ip_model_dict}') + ip_model_dict['faceid_embeds'] = embeds + images = ip_model.generate(**ip_model_dict) + + processed = processing.Processed( + p, + images_list=images, + seed=p.seed, + subseed=p.subseed, + index_of_first_image=0, + ) + processed.infotexts = processed.infotext(p, 0) + ip_model = None + devices.torch_gc() + return processed