Stable Diffusion Image Processing Code
Stable Diffusion Image Processing Code
import json
import logging
import math
import os
import sys
import hashlib
from dataclasses import dataclass, field
import torch
import numpy as np
from PIL import Image, ImageOps
import random
import cv2
from skimage import exposure
from typing import Any
import modules.sd_hijack
from modules import devices, prompt_parser, masking, sd_samplers, lowvram,
infotext_utils, extra_networks, sd_vae_approx, scripts, sd_samplers_common,
sd_unet, errors, rng, profiling
from [Link] import slerp, get_noise_source_type # noqa: F401
from modules.sd_samplers_common import images_tensor_to_samples,
decode_first_stage, approximation_indexes
from [Link] import opts, cmd_opts, state
from [Link] import set_config
import [Link] as shared
import [Link] as paths
import modules.face_restoration
import [Link] as images
import [Link]
import modules.sd_models as sd_models
import modules.sd_vae as sd_vae
# some of those options should not be changed at all because they would break the
model, so I removed them from options.
opt_C = 4
opt_f = 8
def setup_color_correction(image):
[Link]("Calibrating color correction.")
correction_target = [Link]([Link]([Link]()), cv2.COLOR_RGB2LAB)
return correction_target
return [Link]('RGB')
return image
original_denoised_image = [Link]()
image = [Link]('RGBA')
image.alpha_composite(overlay)
image = [Link]('RGB')
# The "masked-image" in this case will just be all 0.5 since the entire
image is masked.
image_conditioning = [Link]([Link][0], 3, height, width,
device=[Link]) * 0.5
image_conditioning = images_tensor_to_samples(image_conditioning,
approximation_indexes.get(opts.sd_vae_encode_method))
return image_conditioning
else:
# Dummy zero conditioning if we're not using inpainting or unclip models.
# Still takes up a bit of memory, but no encoder call.
# Pretty sure we can just make this a 1x1 image since its not going to be
used besides its batch size.
return x.new_zeros([Link][0], 5, 1, 1, dtype=[Link], device=[Link])
@dataclass(repr=False)
class StableDiffusionProcessing:
sd_model: object = None
outpath_samples: str = None
outpath_grids: str = None
prompt: str = ""
prompt_for_display: str = None
negative_prompt: str = ""
styles: list[str] = None
seed: int = -1
subseed: int = -1
subseed_strength: float = 0
seed_resize_from_h: int = -1
seed_resize_from_w: int = -1
seed_enable_extras: bool = True
sampler_name: str = None
scheduler: str = None
batch_size: int = 1
n_iter: int = 1
steps: int = 50
cfg_scale: float = 7.0
distilled_cfg_scale: float = 3.5
width: int = 512
height: int = 512
restore_faces: bool = None
tiling: bool = None
do_not_save_samples: bool = False
do_not_save_grid: bool = False
extra_generation_params: dict[str, Any] = None
overlay_images: list = None
eta: float = None
do_not_reload_embeddings: bool = False
denoising_strength: float = None
ddim_discretize: str = None
s_min_uncond: float = None
s_churn: float = None
s_tmax: float = None
s_tmin: float = None
s_noise: float = None
override_settings: dict[str, Any] = None
override_settings_restore_afterwards: bool = True
sampler_index: int = None
refiner_checkpoint: str = None
refiner_switch_at: float = None
token_merging_ratio = 0
token_merging_ratio_hr = 0
disable_extra_networks: bool = False
firstpass_image: Image = None
scripts_value: [Link] = field(default=None, init=False)
script_args_value: list = field(default=None, init=False)
scripts_setup_complete: bool = field(default=False, init=False)
latents_after_sampling = []
pixels_after_sampling = []
def clear_prompt_cache(self):
self.cached_c = [None, None, None]
self.cached_uc = [None, None, None]
StableDiffusionProcessing.cached_c = [None, None, None]
StableDiffusionProcessing.cached_uc = [None, None, None]
def __post_init__(self):
if self.sampler_index is not None:
print("sampler_index argument for StableDiffusionProcessing does not do
anything; use sampler_name", file=[Link])
[Link] = {}
if [Link] is None:
[Link] = []
self.sampler_noise_scheduler_override = None
self.extra_generation_params = self.extra_generation_params or {}
self.override_settings = self.override_settings or {}
self.script_args = self.script_args or {}
self.refiner_checkpoint_info = None
if not self.seed_enable_extras:
[Link] = -1
self.subseed_strength = 0
self.seed_resize_from_h = 0
self.seed_resize_from_w = 0
self.cached_uc = StableDiffusionProcessing.cached_uc
self.cached_c = StableDiffusionProcessing.cached_c
self.extra_result_images = []
self.latents_after_sampling = []
self.pixels_after_sampling = []
self.modified_noise = None
def fill_fields_from_opts(self):
self.s_min_uncond = self.s_min_uncond if self.s_min_uncond is not None else
opts.s_min_uncond
self.s_churn = self.s_churn if self.s_churn is not None else opts.s_churn
self.s_tmin = self.s_tmin if self.s_tmin is not None else opts.s_tmin
self.s_tmax = (self.s_tmax if self.s_tmax is not None else opts.s_tmax) or
float('inf')
self.s_noise = self.s_noise if self.s_noise is not None else opts.s_noise
@property
def sd_model(self):
return shared.sd_model
@sd_model.setter
def sd_model(self, value):
pass
@property
def scripts(self):
return self.scripts_value
@[Link]
def scripts(self, value):
self.scripts_value = value
@property
def script_args(self):
return self.script_args_value
@script_args.setter
def script_args(self, value):
self.script_args_value = value
def setup_scripts(self):
self.scripts_setup_complete = True
return conditioning_image
if round_image_mask:
# Caller is requesting a discretized mask as input, so we round
to either 1.0 or 0.0
conditioning_mask = [Link](conditioning_mask)
else:
conditioning_mask = source_image.new_ones(1, 1, *source_image.shape[-
2:])
# Create another latent image, this time with a masked version of the
original input.
# Smoothly interpolate between the masked and unmasked latent conditioning
image using a parameter.
conditioning_mask = conditioning_mask.to(device=source_image.device,
dtype=source_image.dtype)
conditioning_image = [Link](
source_image,
source_image * (1.0 - conditioning_mask),
getattr(self, "inpainting_mask_weight",
[Link].inpainting_mask_weight)
)
return image_conditioning
# if self.sd_model.cond_stage_key == "edit":
# return self.edit_image_conditioning(source_image)
if self.sd_model.is_inpaint:
return self.inpainting_image_conditioning(source_image, latent_image,
image_mask=image_mask, round_image_mask=round_image_mask)
# if [Link].conditioning_key == "crossattn-adm":
# return self.unclip_image_conditioning(source_image)
#
# if [Link].model_wrap.inner_model.is_sdxl_inpaint:
# return self.inpainting_image_conditioning(source_image, latent_image,
image_mask=image_mask)
def close(self):
[Link] = None
self.c = None
[Link] = None
if not opts.persistent_cond_cache:
StableDiffusionProcessing.cached_c = [None, None]
StableDiffusionProcessing.cached_uc = [None, None]
def setup_prompts(self):
if isinstance([Link],list):
self.all_prompts = [Link]
elif isinstance(self.negative_prompt, list):
self.all_prompts = [[Link]] * len(self.negative_prompt)
else:
self.all_prompts = self.batch_size * self.n_iter * [[Link]]
if isinstance(self.negative_prompt, list):
self.all_negative_prompts = self.negative_prompt
else:
self.all_negative_prompts = [self.negative_prompt] *
len(self.all_prompts)
if len(self.all_prompts) != len(self.all_negative_prompts):
raise RuntimeError(f"Received a different number of prompts
({len(self.all_prompts)}) and negative prompts ({len(self.all_negative_prompts)})")
self.all_prompts = [shared.prompt_styles.apply_styles_to_prompt(x,
[Link]) for x in self.all_prompts]
self.all_negative_prompts =
[shared.prompt_styles.apply_negative_styles_to_prompt(x, [Link]) for x in
self.all_negative_prompts]
self.main_prompt = self.all_prompts[0]
self.main_negative_prompt = self.all_negative_prompts[0]
return (
required_prompts,
self.distilled_cfg_scale,
self.hr_distilled_cfg,
steps,
hires_steps,
use_old_scheduling,
opts.CLIP_stop_at_last_layers,
shared.sd_model.sd_checkpoint_info,
extra_network_data,
opts.sdxl_crop_left,
opts.sdxl_crop_top,
[Link],
[Link],
opts.fp8_storage,
opts.cache_fp16_weight,
[Link],
)
if [Link].use_old_scheduling:
old_schedules =
prompt_parser.get_learned_conditioning_prompt_schedules(required_prompts, steps,
hires_steps, False)
new_schedules =
prompt_parser.get_learned_conditioning_prompt_schedules(required_prompts, steps,
hires_steps, True)
if old_schedules != new_schedules:
self.extra_generation_params["Old prompt editing timelines"] = True
cache = caches[0]
with [Link]():
shared.sd_model.set_clip_skip(int(opts.CLIP_stop_at_last_layers))
import backend.text_processing.classic_engine
last_extra_generation_params =
backend.text_processing.classic_engine.last_extra_generation_params.copy()
shared.sd_model.extra_generation_params.update(last_extra_generation_params)
if len(cache) > 2:
cache[2] = last_extra_generation_params
backend.text_processing.classic_engine.last_extra_generation_params =
{}
cache[0] = cached_params
return cache[1]
def setup_conds(self):
prompts = prompt_parser.SdConditioning([Link], width=[Link],
height=[Link], distilled_cfg_scale=self.distilled_cfg_scale)
negative_prompts = prompt_parser.SdConditioning(self.negative_prompts,
width=[Link], height=[Link], is_negative_prompt=True,
distilled_cfg_scale=self.distilled_cfg_scale)
sampler_config = sd_samplers.find_sampler_config(self.sampler_name)
total_steps = sampler_config.total_steps([Link]) if sampler_config else
[Link]
self.step_multiplier = total_steps // [Link]
self.firstpass_steps = total_steps
if self.cfg_scale == 1:
[Link] = None
print('Skipping unconditional conditioning when CFG = 1. Negative
Prompts are ignored.')
else:
[Link] =
self.get_conds_with_caching(prompt_parser.get_learned_conditioning,
negative_prompts, total_steps, [self.cached_uc], self.extra_network_data)
self.c =
self.get_conds_with_caching(prompt_parser.get_multicond_learned_conditioning,
prompts, total_steps, [self.cached_c], self.extra_network_data)
def get_conds(self):
return self.c, [Link]
def parse_extra_network_prompts(self):
[Link], self.extra_network_data =
extra_networks.parse_prompts([Link])
class Processed:
def __init__(self, p: StableDiffusionProcessing, images_list, seed=-1, info="",
subseed=None, all_prompts=None, all_negative_prompts=None, all_seeds=None,
all_subseeds=None, index_of_first_image=0, infotexts=None, comments="",
extra_images_list=[]):
[Link] = images_list
self.extra_images = extra_images_list
[Link] = [Link]
self.negative_prompt = p.negative_prompt
[Link] = seed
[Link] = subseed
self.subseed_strength = p.subseed_strength
[Link] = info
[Link] = "".join(f"{comment}\n" for comment in [Link])
[Link] = [Link]
[Link] = [Link]
self.sampler_name = p.sampler_name
self.cfg_scale = p.cfg_scale
self.image_cfg_scale = getattr(p, 'image_cfg_scale', None)
[Link] = [Link]
self.batch_size = p.batch_size
self.restore_faces = p.restore_faces
self.face_restoration_model = opts.face_restoration_model if
p.restore_faces else None
self.sd_model_name = p.sd_model_name
self.sd_model_hash = p.sd_model_hash
self.sd_vae_name = p.sd_vae_name
self.sd_vae_hash = p.sd_vae_hash
self.seed_resize_from_w = p.seed_resize_from_w
self.seed_resize_from_h = p.seed_resize_from_h
self.denoising_strength = getattr(p, 'denoising_strength', None)
self.extra_generation_params = p.extra_generation_params
self.index_of_first_image = index_of_first_image
[Link] = [Link]
self.job_timestamp = state.job_timestamp
self.clip_skip = int(opts.CLIP_stop_at_last_layers)
self.token_merging_ratio = p.token_merging_ratio
self.token_merging_ratio_hr = p.token_merging_ratio_hr
[Link] = [Link]
self.ddim_discretize = p.ddim_discretize
self.s_churn = p.s_churn
self.s_tmin = p.s_tmin
self.s_tmax = p.s_tmax
self.s_noise = p.s_noise
self.s_min_uncond = p.s_min_uncond
self.sampler_noise_scheduler_override = p.sampler_noise_scheduler_override
[Link] = [Link] if not isinstance([Link], list) else
[Link][0]
self.negative_prompt = self.negative_prompt if not
isinstance(self.negative_prompt, list) else self.negative_prompt[0]
[Link] = int([Link] if not isinstance([Link], list) else
[Link][0]) if [Link] is not None else -1
[Link] = int([Link] if not isinstance([Link], list) else
[Link][0]) if [Link] is not None else -1
self.is_using_inpainting_conditioning = p.is_using_inpainting_conditioning
def js(self):
obj = {
"prompt": self.all_prompts[0],
"all_prompts": self.all_prompts,
"negative_prompt": self.all_negative_prompts[0],
"all_negative_prompts": self.all_negative_prompts,
"seed": [Link],
"all_seeds": self.all_seeds,
"subseed": [Link],
"all_subseeds": self.all_subseeds,
"subseed_strength": self.subseed_strength,
"width": [Link],
"height": [Link],
"sampler_name": self.sampler_name,
"cfg_scale": self.cfg_scale,
"steps": [Link],
"batch_size": self.batch_size,
"restore_faces": self.restore_faces,
"face_restoration_model": self.face_restoration_model,
"sd_model_name": self.sd_model_name,
"sd_model_hash": self.sd_model_hash,
"sd_vae_name": self.sd_vae_name,
"sd_vae_hash": self.sd_vae_hash,
"seed_resize_from_w": self.seed_resize_from_w,
"seed_resize_from_h": self.seed_resize_from_h,
"denoising_strength": self.denoising_strength,
"extra_generation_params": self.extra_generation_params,
"index_of_first_image": self.index_of_first_image,
"infotexts": [Link],
"styles": [Link],
"job_timestamp": self.job_timestamp,
"clip_skip": self.clip_skip,
"is_using_inpainting_conditioning":
self.is_using_inpainting_conditioning,
"version": [Link],
}
class DecodedSamples(list):
already_decoded = True
for x in samples_pytorch:
[Link](x)
return samples
def get_fixed_seed(seed):
if seed == '' or seed is None:
seed = -1
elif isinstance(seed, str):
try:
seed = int(seed)
except Exception:
seed = -1
if seed == -1:
return int([Link](4294967294))
return seed
def fix_seed(p):
[Link] = get_fixed_seed([Link])
[Link] = get_fixed_seed([Link])
def program_version():
import launch
res = launch.git_tag()
if res == "<none>":
res = None
return res
Returns: str
Extra generation params
p.extra_generation_params dictionary allows for additional parameters to be
added to the infotext
this can be use by the base webui or extensions.
To add a new entry, add a new key value pair, the dictionary key will be used
as the key of the parameter in the infotext
the value generation_params can be defined as:
- str | None
- List[str|None]
- callable func(**kwargs) -> str | None
Defining as a list allows for parameter that changes across images in the job,
for example, the 'Seed' parameter.
The list should have the same length as the total number of images in the
entire job.
The function takes locals() as **kwargs, as such will have access to variables
like 'p' and 'index'.
the base signature of the function should be:
func(**kwargs) -> str | None
optionally it can have additional arguments that will be used in the function:
func(p, index, **kwargs) -> str | None
note: for better future compatibility even though this function will have
access to all variables in the locals(),
it is recommended to only use the arguments present in the function
signature of create_infotext.
For actual implementation examples, see [Link] >
get_hr_prompt.
"""
if use_main_prompt:
index = 0
elif index is None:
index = position_in_batch + iteration * p.batch_size
if all_negative_prompts is None:
all_negative_prompts = p.all_negative_prompts
uses_ensd = opts.eta_noise_seed_delta != 0
if uses_ensd:
uses_ensd = sd_samplers_common.is_sampler_using_eta_noise_seed_delta(p)
generation_params = {
"Steps": [Link],
"Sampler": p.sampler_name,
"Schedule type": [Link],
"CFG scale": p.cfg_scale
}
noise_source_type = get_noise_source_type()
generation_params.update({
"Image CFG scale": getattr(p, 'image_cfg_scale', None),
"Seed": p.all_seeds[0] if use_main_prompt else all_seeds[index],
"Face restoration": opts.face_restoration_model if p.restore_faces else
None,
"Size": f"{[Link]}x{[Link]}",
"Model hash": p.sd_model_hash if opts.add_model_hash_to_info else None,
"Model": p.sd_model_name if opts.add_model_name_to_info else None,
"FP8 weight": opts.fp8_storage if devices.fp8 else None,
"Cache FP16 weight for LoRA": opts.cache_fp16_weight if devices.fp8 else
None,
# "VAE hash": p.sd_vae_hash if opts.add_vae_hash_to_info else None,
# "VAE": p.sd_vae_name if opts.add_vae_name_to_info else None,
"Variation seed": (None if p.subseed_strength == 0 else (p.all_subseeds[0]
if use_main_prompt else all_subseeds[index])),
"Variation seed strength": (None if p.subseed_strength == 0 else
p.subseed_strength),
"Seed resize from": (None if p.seed_resize_from_w <= 0 or
p.seed_resize_from_h <= 0 else f"{p.seed_resize_from_w}x{p.seed_resize_from_h}"),
"Denoising strength": p.extra_generation_params.get("Denoising strength"),
"Conditional mask weight": getattr(p, "inpainting_mask_weight",
[Link].inpainting_mask_weight) if p.is_using_inpainting_conditioning else
None,
"Clip skip": None if clip_skip <= 1 else clip_skip,
"ENSD": opts.eta_noise_seed_delta if uses_ensd else None,
"Token merging ratio": None if token_merging_ratio == 0 else
token_merging_ratio,
"Token merging ratio hr": None if not enable_hr or token_merging_ratio_hr
== 0 else token_merging_ratio_hr,
"Init image hash": getattr(p, 'init_img_hash', None),
"RNG": noise_source_type if noise_source_type != "GPU" else None,
"Tiling": "True" if [Link] else None,
**p.extra_generation_params,
"Version": program_version() if opts.add_version_to_infotext else None,
"User": [Link] if opts.add_user_name_to_info else None,
})
if [Link].forge_unet_storage_dtype != 'Automatic':
generation_params['Diffusion in Low Bits'] =
[Link].forge_unet_storage_dtype
if isinstance([Link].forge_additional_modules, list) and
len([Link].forge_additional_modules) > 0:
for i, m in enumerate([Link].forge_additional_modules):
generation_params[f'Module {i+1}'] =
[Link]([Link](m))[0]
return f"{prompt_text}{negative_prompt_text}\n{generation_params_text}".strip()
need_global_unload = False
if need_global_unload:
p.clear_prompt_cache()
need_global_unload = False
try:
# if no checkpoint override or the override checkpoint can't be found,
remove override entry and load opts checkpoint
# and if after running refiner, the refiner model is not unloaded - webui
swaps back to main model here, if model over is present it will be reloaded
afterwards
if
sd_models.checkpoint_aliases.get(p.override_settings.get('sd_model_checkpoint')) is
None:
p.override_settings.pop('sd_model_checkpoint', None)
with [Link]():
res = process_images_inner(p)
finally:
# restore original options
if p.override_settings_restore_afterwards:
set_config(stored_opts, save_config=False)
return res
if isinstance([Link], list):
assert(len([Link]) > 0)
else:
assert [Link] is not None
devices.torch_gc()
seed = get_fixed_seed([Link])
subseed = get_fixed_seed([Link])
if p.restore_faces is None:
p.restore_faces = opts.face_restoration
if [Link] is None:
[Link] = [Link]
if hasattr(shared.sd_model, 'fix_dimensions'):
[Link], [Link] = shared.sd_model.fix_dimensions([Link], [Link])
p.sd_model_name = shared.sd_model.sd_checkpoint_info.name_for_extra
p.sd_model_hash = shared.sd_model.sd_model_hash
p.sd_vae_name = sd_vae.get_loaded_vae_name()
p.sd_vae_hash = sd_vae.get_loaded_vae_hash()
apply_circular_forge(p.sd_model, [Link])
p.sd_model.comments = []
p.sd_model.extra_generation_params = {}
p.fill_fields_from_opts()
p.setup_prompts()
if isinstance(seed, list):
p.all_seeds = seed
else:
p.all_seeds = [int(seed) + (x if p.subseed_strength == 0 else 0) for x in
range(len(p.all_prompts))]
if isinstance(subseed, list):
p.all_subseeds = subseed
else:
p.all_subseeds = [int(subseed) + x for x in range(len(p.all_prompts))]
infotexts = []
output_images = []
with torch.inference_mode():
with [Link]():
[Link](p.all_prompts, p.all_seeds, p.all_subseeds)
# for OSX, loading the model during sampling changes the generated
picture, so it is loaded here
if [Link].live_previews_enable and opts.show_progress_type ==
"Approx NN":
sd_vae_approx.model()
sd_unet.apply_unet()
if state.job_count == -1:
state.job_count = p.n_iter
for n in range(p.n_iter):
[Link] = n
if [Link]:
[Link] = False
if [Link] or state.stopping_generation:
break
p.sd_model.forge_objects =
p.sd_model.forge_objects_original.shallow_copy()
[Link] = p.all_prompts[n * p.batch_size:(n + 1) * p.batch_size]
p.negative_prompts = p.all_negative_prompts[n * p.batch_size:(n + 1) *
p.batch_size]
[Link] = p.all_seeds[n * p.batch_size:(n + 1) * p.batch_size]
[Link] = p.all_subseeds[n * p.batch_size:(n + 1) * p.batch_size]
latent_channels = shared.sd_model.forge_objects.vae.latent_channels
[Link] = [Link]((latent_channels, [Link] // opt_f, [Link] //
opt_f), [Link], subseeds=[Link], subseed_strength=p.subseed_strength,
seed_resize_from_h=p.seed_resize_from_h, seed_resize_from_w=p.seed_resize_from_w)
if len([Link]) == 0:
break
p.parse_extra_network_prompts()
if not p.disable_extra_networks:
extra_networks.activate(p, p.extra_network_data)
p.sd_model.forge_objects =
p.sd_model.forge_objects_after_applying_lora.shallow_copy()
p.setup_conds()
p.extra_generation_params.update(p.sd_model.extra_generation_params)
if p.n_iter > 1:
[Link] = f"Batch {n+1} out of {p.n_iter}"
sigmas_backup = None
if (opts.sd_noise_schedule == "Zero Terminal SNR" or
(hasattr(p.sd_model.model_config, 'ztsnr') and p.sd_model.model_config.ztsnr)) and
p is not None:
p.extra_generation_params['Noise Schedule'] =
opts.sd_noise_schedule
sigmas_backup =
p.sd_model.forge_objects.[Link]
p.sd_model.forge_objects.[Link].set_sigmas(rescale_zero_terminal_snr_
sigmas(p.sd_model.forge_objects.[Link]))
samples_ddim = [Link](conditioning=p.c,
unconditional_conditioning=[Link], seeds=[Link], subseeds=[Link],
subseed_strength=p.subseed_strength, prompts=[Link])
p.sd_model.forge_objects.[Link].set_sigmas(sigmas_backup)
if opts.sd_vae_decode_method != 'Full':
p.extra_generation_params['VAE Decoder'] =
opts.sd_vae_decode_method
x_samples_ddim = decode_latent_batch(p.sd_model, samples_ddim,
target_device=[Link], check_for_nans=True)
x_samples_ddim = [Link](x_samples_ddim).float()
x_samples_ddim = [Link]((x_samples_ddim + 1.0) / 2.0, min=0.0,
max=1.0)
del samples_ddim
devices.torch_gc()
[Link]()
batch_params =
[Link](list(x_samples_ddim))
[Link].postprocess_batch_list(p, batch_params, batch_number=n)
x_samples_ddim = batch_params.images
def infotext(index=0, use_main_prompt=False):
return create_infotext(p, [Link], [Link], [Link],
use_main_prompt=use_main_prompt, index=index,
all_negative_prompts=p.negative_prompts)
save_samples = p.save_samples()
if p.restore_faces:
if save_samples and opts.save_images_before_face_restoration:
images.save_image([Link](x_sample),
p.outpath_samples, "", [Link][i], [Link][i], opts.samples_format,
info=infotext(i), p=p, suffix="-before-face-restoration")
devices.torch_gc()
x_sample = modules.face_restoration.restore_faces(x_sample)
devices.torch_gc()
image = [Link](x_sample)
if not [Link].overlay_inpaint:
overlay_image = None
elif getattr(p, "overlay_images", None) is not None and i <
len(p.overlay_images):
overlay_image = p.overlay_images[i]
else:
overlay_image = None
p.pixels_after_sampling.append(image)
if save_samples:
images.save_image(image, p.outpath_samples, "", [Link][i],
[Link][i], opts.samples_format, info=infotext(i), p=p)
text = infotext(i)
[Link](text)
if opts.enable_pnginfo:
[Link]["parameters"] = text
output_images.append(image)
if opts.return_mask_composite or opts.save_mask_composite:
image_mask_composite =
[Link](original_denoised_image.convert('RGBA').convert('RGBa'),
[Link]('RGBa', [Link]), images.resize_image(2, mask_for_overlay,
[Link], [Link]).convert('L')).convert('RGBA')
if save_samples and opts.save_mask_composite:
images.save_image(image_mask_composite,
p.outpath_samples, "", [Link][i], [Link][i], opts.samples_format,
info=infotext(i), p=p, suffix="-mask-composite")
if opts.return_mask_composite:
output_images.append(image_mask_composite)
del x_samples_ddim
devices.torch_gc()
if not infotexts:
[Link](Processed(p, []).infotext(p, 0))
p.color_corrections = None
index_of_first_image = 0
unwanted_grid_because_of_img_count = len(output_images) < 2 and
opts.grid_only_if_multiple
if (opts.return_grid or opts.grid_save) and not p.do_not_save_grid and not
unwanted_grid_because_of_img_count:
grid = images.image_grid(output_images, p.batch_size)
if opts.return_grid:
text = infotext(use_main_prompt=True)
[Link](0, text)
if opts.enable_pnginfo:
[Link]["parameters"] = text
output_images.insert(0, grid)
index_of_first_image = 1
if opts.grid_save:
images.save_image(grid, p.outpath_grids, "grid", p.all_seeds[0],
p.all_prompts[0], opts.grid_format, info=infotext(use_main_prompt=True),
short_filename=not opts.grid_extended_filename, p=p, grid=True)
devices.torch_gc()
res = Processed(
p,
images_list=output_images,
seed=p.all_seeds[0],
info=infotexts[0],
subseed=p.all_subseeds[0],
index_of_first_image=index_of_first_image,
infotexts=infotexts,
extra_images_list=p.extra_result_images,
)
return res
def process_extra_images(processed:Processed):
"""used by API processing functions to ensure extra images are PIL image
objects"""
extra_images = []
for img in processed.extra_images:
if isinstance(img, [Link]):
img = [Link](img)
if not [Link](img):
continue
extra_images.append(img)
processed.extra_images = extra_images
@dataclass(repr=False)
class StableDiffusionProcessingTxt2Img(StableDiffusionProcessing):
enable_hr: bool = False
denoising_strength: float = 0.75
firstphase_width: int = 0
firstphase_height: int = 0
hr_scale: float = 2.0
hr_upscaler: str = None
hr_second_pass_steps: int = 0
hr_resize_x: int = 0
hr_resize_y: int = 0
hr_checkpoint_name: str = None
hr_additional_modules: list = field(default=None)
hr_sampler_name: str = None
hr_scheduler: str = None
hr_prompt: str = ''
hr_negative_prompt: str = ''
hr_cfg: float = 1.0
hr_distilled_cfg: float = 3.5
force_task_id: str = None
def __post_init__(self):
super().__post_init__()
if self.firstphase_width != 0 or self.firstphase_height != 0:
self.hr_upscale_to_x = [Link]
self.hr_upscale_to_y = [Link]
[Link] = self.firstphase_width
[Link] = self.firstphase_height
self.cached_hr_uc = StableDiffusionProcessingTxt2Img.cached_hr_uc
self.cached_hr_c = StableDiffusionProcessingTxt2Img.cached_hr_c
def calculate_target_resolution(self):
if opts.use_old_hires_fix_width_height and
self.applied_old_hires_behavior_to != ([Link], [Link]):
self.hr_resize_x = [Link]
self.hr_resize_y = [Link]
self.hr_upscale_to_x = [Link]
self.hr_upscale_to_y = [Link]
[Link], [Link] =
old_hires_fix_first_pass_dimensions([Link], [Link])
self.applied_old_hires_behavior_to = ([Link], [Link])
if self.hr_resize_y == 0:
self.hr_upscale_to_x = self.hr_resize_x
self.hr_upscale_to_y = self.hr_resize_x * [Link] // [Link]
elif self.hr_resize_x == 0:
self.hr_upscale_to_x = self.hr_resize_y * [Link] // [Link]
self.hr_upscale_to_y = self.hr_resize_y
else:
target_w = self.hr_resize_x
target_h = self.hr_resize_y
src_ratio = [Link] / [Link]
dst_ratio = self.hr_resize_x / self.hr_resize_y
if self.hr_checkpoint_info is None:
raise Exception(f'Could not find checkpoint with name
{self.hr_checkpoint_name}')
self.extra_generation_params["Hires checkpoint"] =
self.hr_checkpoint_info.short_title
if isinstance(self.hr_additional_modules, list):
if self.hr_additional_modules == []:
self.extra_generation_params['Hires Module 1'] = 'Built-in'
elif 'Use same choices' in self.hr_additional_modules:
self.extra_generation_params['Hires Module 1'] = 'Use same
choices'
else:
for i, m in enumerate(self.hr_additional_modules):
self.extra_generation_params[f'Hires Module {i+1}'] =
[Link]([Link](m))[0]
if self.hr_scheduler is None:
self.hr_scheduler = [Link]
self.latent_scale_mode =
shared.latent_upscale_modes.get(self.hr_upscaler, None) if self.hr_upscaler is not
None else shared.latent_upscale_modes.get(shared.latent_upscale_default_mode,
"nearest")
if self.enable_hr and self.latent_scale_mode is None:
if not any([Link] == self.hr_upscaler for x in
shared.sd_upscalers):
raise Exception(f"could not find upscaler named
{self.hr_upscaler}")
self.calculate_target_resolution()
if not state.processing_has_refined_job_count:
if state.job_count == -1:
state.job_count = self.n_iter
if getattr(self, 'txt2img_upscale', False):
total_steps = (self.hr_second_pass_steps or [Link]) *
state.job_count
else:
total_steps = ([Link] + (self.hr_second_pass_steps or
[Link])) * state.job_count
shared.total_tqdm.updateTotal(total_steps)
state.job_count = state.job_count * 2
state.processing_has_refined_job_count = True
if self.hr_second_pass_steps:
self.extra_generation_params["Hires steps"] =
self.hr_second_pass_steps
if self.latent_scale_mode is None:
image = [Link](self.firstpass_image).astype(np.float32) / 255.0 *
2.0 - 1.0
image = [Link](image, 2, 0)
samples = None
decoded_samples = [Link](np.expand_dims(image, 0))
else:
image = [Link](self.firstpass_image).astype(np.float32) / 255.0
image = [Link](image, 2, 0)
image = torch.from_numpy(np.expand_dims(image, axis=0))
image = [Link]([Link], dtype=torch.float32)
if opts.sd_vae_encode_method != 'Full':
self.extra_generation_params['VAE Encoder'] =
opts.sd_vae_encode_method
samples = images_tensor_to_samples(image,
approximation_indexes.get(opts.sd_vae_encode_method), self.sd_model)
decoded_samples = None
devices.torch_gc()
else:
# here we generate an image normally
x = [Link]()
self.sd_model.forge_objects =
self.sd_model.forge_objects_after_applying_lora.shallow_copy()
apply_token_merging(self.sd_model, self.get_token_merging_ratio())
uc=unconditional_conditioning)
if self.modified_noise is not None:
x = self.modified_noise
self.modified_noise = None
if not self.enable_hr:
return samples
devices.torch_gc()
if self.latent_scale_mode is None:
decoded_samples = [Link](decode_latent_batch(self.sd_model,
samples, target_device=[Link], check_for_nans=True)).to(dtype=torch.float32)
else:
decoded_samples = None
with sd_models.SkipWritingToConfig():
fp_checkpoint = getattr([Link], 'sd_model_checkpoint')
fp_additional_modules = getattr([Link],
'forge_additional_modules')
reload = False
if hasattr(self, 'hr_additional_modules') and 'Use same choices' not in
self.hr_additional_modules:
modules_changed =
main_entry.modules_change(self.hr_additional_modules, save=False, refresh=False)
if modules_changed:
reload = True
if reload:
try:
main_entry.refresh_model_loading_parameters()
sd_models.forge_model_reload()
finally:
main_entry.modules_change(fp_additional_modules, save=False,
refresh=False)
main_entry.checkpoint_change(fp_checkpoint, save=False,
refresh=False)
main_entry.refresh_model_loading_parameters()
if self.sd_model.use_distilled_cfg_scale:
self.extra_generation_params['Hires Distilled CFG Scale'] =
self.hr_distilled_cfg
self.is_hr_pass = True
target_width = self.hr_upscale_to_x
target_height = self.hr_upscale_to_y
[Link] = sd_samplers.create_sampler(img2img_sampler_name,
self.sd_model)
samples = [Link](samples,
size=(target_height // opt_f, target_width // opt_f),
mode=self.latent_scale_mode["mode"], antialias=self.latent_scale_mode["antialias"])
batch_images = []
for i, x_sample in enumerate(lowres_samples):
x_sample = 255. * [Link](x_sample.cpu().numpy(), 0, 2)
x_sample = x_sample.astype(np.uint8)
image = [Link](x_sample)
save_intermediate(image, i)
image = images.resize_image(0, image, target_width, target_height,
upscaler_name=self.hr_upscaler)
image = [Link](image).astype(np.float32) / 255.0
image = [Link](image, 2, 0)
batch_images.append(image)
decoded_samples = torch.from_numpy([Link](batch_images))
decoded_samples = decoded_samples.to([Link],
dtype=torch.float32)
if opts.sd_vae_encode_method != 'Full':
self.extra_generation_params['VAE Encoder'] =
opts.sd_vae_encode_method
samples = images_tensor_to_samples(decoded_samples,
approximation_indexes.get(opts.sd_vae_encode_method))
image_conditioning = self.img2img_image_conditioning(decoded_samples,
samples)
[Link]()
# GC now before running the next img2img to prevent running out of memory
devices.torch_gc()
if not self.disable_extra_networks:
with [Link]():
extra_networks.activate(self, self.hr_extra_network_data)
with [Link]():
self.calculate_hr_conds()
self.sd_model.forge_objects =
self.sd_model.forge_objects_after_applying_lora.shallow_copy()
apply_token_merging(self.sd_model,
self.get_token_merging_ratio(for_hr=True))
[Link] = None
devices.torch_gc()
self.is_hr_pass = False
return decoded_samples
def close(self):
super().close()
self.hr_c = None
self.hr_uc = None
if not opts.persistent_cond_cache:
StableDiffusionProcessingTxt2Img.cached_hr_uc = [None, None]
StableDiffusionProcessingTxt2Img.cached_hr_c = [None, None]
def setup_prompts(self):
super().setup_prompts()
if not self.enable_hr:
return
if self.hr_prompt == '':
self.hr_prompt = [Link]
if self.hr_negative_prompt == '':
self.hr_negative_prompt = self.negative_prompt
if isinstance(self.hr_prompt, list):
self.all_hr_prompts = self.hr_prompt
else:
self.all_hr_prompts = self.batch_size * self.n_iter * [self.hr_prompt]
if isinstance(self.hr_negative_prompt, list):
self.all_hr_negative_prompts = self.hr_negative_prompt
else:
self.all_hr_negative_prompts = self.batch_size * self.n_iter *
[self.hr_negative_prompt]
self.all_hr_prompts = [shared.prompt_styles.apply_styles_to_prompt(x,
[Link]) for x in self.all_hr_prompts]
self.all_hr_negative_prompts =
[shared.prompt_styles.apply_negative_styles_to_prompt(x, [Link]) for x in
self.all_hr_negative_prompts]
def calculate_hr_conds(self):
if self.hr_c is not None:
return
hr_prompts = prompt_parser.SdConditioning(self.hr_prompts,
width=self.hr_upscale_to_x, height=self.hr_upscale_to_y,
distilled_cfg_scale=self.hr_distilled_cfg)
hr_negative_prompts =
prompt_parser.SdConditioning(self.hr_negative_prompts, width=self.hr_upscale_to_x,
height=self.hr_upscale_to_y, is_negative_prompt=True,
distilled_cfg_scale=self.hr_distilled_cfg)
sampler_config = sd_samplers.find_sampler_config(self.hr_sampler_name or
self.sampler_name)
steps = self.hr_second_pass_steps or [Link]
total_steps = sampler_config.total_steps(steps) if sampler_config else
steps
if self.hr_cfg == 1:
self.hr_uc = None
print('Skipping unconditional conditioning (HR pass) when CFG = 1.
Negative Prompts are ignored.')
else:
self.hr_uc =
self.get_conds_with_caching(prompt_parser.get_learned_conditioning,
hr_negative_prompts, self.firstpass_steps, [self.cached_hr_uc, self.cached_uc],
self.hr_extra_network_data, total_steps)
self.hr_c =
self.get_conds_with_caching(prompt_parser.get_multicond_learned_conditioning,
hr_prompts, self.firstpass_steps, [self.cached_hr_c, self.cached_c],
self.hr_extra_network_data, total_steps)
def setup_conds(self):
if self.is_hr_pass:
# if we are in hr pass right now, the call is being made from the
refiner, and we don't need to setup firstpass cons or switch model
self.hr_c = None
self.calculate_hr_conds()
return
super().setup_conds()
self.hr_uc = None
self.hr_c = None
self.calculate_hr_conds()
with [Link]():
extra_networks.activate(self, self.extra_network_data)
def get_conds(self):
if self.is_hr_pass:
return self.hr_c, self.hr_uc
return super().get_conds()
def parse_extra_network_prompts(self):
res = super().parse_extra_network_prompts()
if self.enable_hr:
self.hr_prompts = self.all_hr_prompts[[Link] * self.batch_size:
([Link] + 1) * self.batch_size]
self.hr_negative_prompts = self.all_hr_negative_prompts[[Link]
* self.batch_size:([Link] + 1) * self.batch_size]
self.hr_prompts, self.hr_extra_network_data =
extra_networks.parse_prompts(self.hr_prompts)
return res
@dataclass(repr=False)
class StableDiffusionProcessingImg2Img(StableDiffusionProcessing):
init_images: list = None
resize_mode: int = 0
denoising_strength: float = 0.75
image_cfg_scale: float = None
mask: Any = None
mask_blur_x: int = 4
mask_blur_y: int = 4
mask_blur: int = None
mask_round: bool = True
inpainting_fill: int = 0
inpaint_full_res: bool = True
inpaint_full_res_padding: int = 0
inpainting_mask_invert: int = 0
initial_noise_multiplier: float = None
latent_mask: Image = None
force_task_id: str = None
def __post_init__(self):
super().__post_init__()
self.image_mask = [Link]
[Link] = None
self.initial_noise_multiplier = opts.initial_noise_multiplier if
self.initial_noise_multiplier is None else self.initial_noise_multiplier
@property
def mask_blur(self):
if self.mask_blur_x == self.mask_blur_y:
return self.mask_blur_x
return None
@mask_blur.setter
def mask_blur(self, value):
if isinstance(value, int):
self.mask_blur_x = value
self.mask_blur_y = value
image_mask = self.image_mask
if self.inpainting_mask_invert:
image_mask = [Link](image_mask)
self.extra_generation_params["Mask mode"] = "Inpaint not masked"
if self.mask_blur_x > 0:
np_mask = [Link](image_mask)
kernel_size = 2 * int(2.5 * self.mask_blur_x + 0.5) + 1
np_mask = [Link](np_mask, (kernel_size, 1),
self.mask_blur_x)
image_mask = [Link](np_mask)
if self.mask_blur_y > 0:
np_mask = [Link](image_mask)
kernel_size = 2 * int(2.5 * self.mask_blur_y + 0.5) + 1
np_mask = [Link](np_mask, (1, kernel_size),
self.mask_blur_y)
image_mask = [Link](np_mask)
if self.inpaint_full_res:
self.mask_for_overlay = image_mask
mask = image_mask.convert('L')
crop_region = masking.get_crop_region_v2(mask,
self.inpaint_full_res_padding)
if crop_region:
crop_region = masking.expand_crop_region(crop_region,
[Link], [Link], [Link], [Link])
x1, y1, x2, y2 = crop_region
mask = [Link](crop_region)
image_mask = images.resize_image(2, mask, [Link],
[Link])
self.paste_to = (x1, y1, x2-x1, y2-y1)
self.extra_generation_params["Inpaint area"] = "Only masked"
self.extra_generation_params["Masked area padding"] =
self.inpaint_full_res_padding
else:
crop_region = None
image_mask = None
self.mask_for_overlay = None
self.inpaint_full_res = False
massage = 'Unable to perform "Inpaint Only mask" because mask
is blank, switch to img2img mode.'
self.sd_model.[Link](massage)
[Link](massage)
else:
image_mask = images.resize_image(self.resize_mode, image_mask,
[Link], [Link])
np_mask = [Link](image_mask)
np_mask = [Link]((np_mask.astype(np.float32)) * 2, 0,
255).astype(np.uint8)
self.mask_for_overlay = [Link](np_mask)
self.overlay_images = []
self.overlay_images.append(image_masked.convert('RGBA'))
if self.inpainting_fill == 0:
self.extra_generation_params["Masked content"] = 'fill'
if add_color_corrections:
self.color_corrections.append(setup_color_correction(image))
[Link](image)
if len(imgs) == 1:
batch_images = np.expand_dims(imgs[0], axis=0).repeat(self.batch_size,
axis=0)
if self.overlay_images is not None:
self.overlay_images = self.overlay_images * self.batch_size
image = torch.from_numpy(batch_images)
image = [Link]([Link], dtype=torch.float32)
if opts.sd_vae_encode_method != 'Full':
self.extra_generation_params['VAE Encoder'] = opts.sd_vae_encode_method
self.init_latent = images_tensor_to_samples(image,
approximation_indexes.get(opts.sd_vae_encode_method), self.sd_model)
devices.torch_gc()
if self.resize_mode == 3:
self.init_latent = [Link](self.init_latent,
size=([Link] // opt_f, [Link] // opt_f), mode="bilinear")
elif self.inpainting_fill == 3:
self.init_latent = self.init_latent * [Link]
self.extra_generation_params["Masked content"] = 'latent nothing'
self.image_conditioning = self.img2img_image_conditioning(image * 2 - 1,
self.init_latent, image_mask, self.mask_round)
if self.initial_noise_multiplier != 1.0:
self.extra_generation_params["Noise multiplier"] =
self.initial_noise_multiplier
x *= self.initial_noise_multiplier
self.sd_model.forge_objects =
self.sd_model.forge_objects_after_applying_lora.shallow_copy()
apply_token_merging(self.sd_model, self.get_token_merging_ratio())
uc=unconditional_conditioning)
samples = blended_samples
del x
devices.torch_gc()
return samples
The document details a structured approach for handling the processing of scripts both before and during image generation, where scripts can prepare the environment and alter parameters. Scripts like 'before_process' and 'before_process_batch' are invoked to set up prompts and manage model caches. The implementation of these processes optimizes efficiency by ensuring that configurations are in place before the computationally intensive tasks of image generation commence, reducing unnecessary recalculations and improving overall workflow .
User options like model restoration and conditional weights significantly influence image generation by providing users with the flexibility to adjust aspects like facial feature regeneration and masking influences. The document describes these settings as being part of a comprehensive parameter set that allows users to tailor outcomes based on specific needs, such as more accurate face depiction or controlled image variations. These options enhance the degree of user control over the output, allowing for precise customization while also posing a challenge in balancing complexity with usability .
The document describes that during the upscaling process, noise levels are calculated using a combination of fixed seeds and noise source types that may differ based on the GPU or specified random generators. This method involves creating RNG objects specifically for the dimensions involved in upscaling and explicitly adjusting for the expected noise during transitions to higher resolutions. It allows for precision in noise estimation, which is crucial in maintaining image quality and detail throughout the upscaling process, especially when larger image dimensions might amplify noise artifacts .
Fixed seed values in the image generation process provide benefits such as repeatability and consistency of results. By using fixed seeds, the same random values are applied, leading to identical outputs given the same initial parameters and conditions, which is useful for testing and reproducibility. However, a limitation is the lack of diversity; using the same seed can result in similar patterns or outputs across images, reducing variability and creativity. This approach may not be suitable for tasks aiming to generate diverse or exploratory content .
The use of a callable function affects parameter generation by allowing for dynamic computation of parameter values at runtime. This method is necessary for parameters that cannot be predetermined or might vary depending on other variables such as 'p' and 'index'. For instance, a static string or list would not adequately handle parameters like 'Hires prompt', which might be altered by processes in the pipeline or extensions and could differ across images. Therefore, using a callable allows for additional logic and access to contextual variables, ensuring more accurate and tailored parameter generation .
Applying 'High Resolution Fix (Hires Fix)' involves several considerations including the potential to improve image quality by scaling initial low-resolution outputs to higher resolutions, while carefully managing noise and artifacts through advanced conditioning techniques. The document specifies the need for compatible prompts and configurations to ensure model integrity is preserved during the upscale. Parameters such as 'Steps' and the use of conditional weighting might be adjusted to accommodate the fix, offering a balance between attained detail and computational load .
The 'face restoration' option is significant in the parameter generation process as it determines whether the image generation model attempts to improve the quality and realism of facial features in the output. This is particularly important because faces are a critical aspect of human visual perception and are more prone to scrutiny. If enabled, the parameter refers to the specific face restoration model specified in options, which enhances the overall quality and appeal of generated images, particularly for applications focusing on photorealistic or high-quality human images .
The 'Init image hash' serves as a unique identifier for initial images in the generation pipeline. Its significance lies in enabling traceability and consistency, allowing for version control and reproducibility of image generation outcomes. By hashing images, the system ensures that the same inputs can be accurately identified and reprocessed, which is critical for debugging, validating experiments, and reducing redundancy in image handling .
The document outlines that the 'hires prompt' is used as an alternative to the 'main prompt' specifically during the high-resolution pass. If 'use_main_prompt' is true, the system defaults to using the main prompt settings; otherwise, it switches to a different index or prompt structure as designated by the user. This allows for flexibility in altering image aspects based on resolution needs without completely discarding the base prompt details, ensuring continuity and specificity of prompts across different processing stages .
In the document's context, 'clip skip' affects the number of layers utilized in the CLIP model during image generation. It only becomes active if the 'clip_skip' value exceeds 1, which suggests an intention to skip certain layers of the CLIP model to potentially improve runtime efficiency or tweak model behavior. This can be useful in optimizing performance without significantly compromising the quality of the generated images .