stable-diffusion-webui/modules/sd_vae_taesd.py

"""
Tiny AutoEncoder for Stable Diffusion
(DNN for encoding / decoding SD's latent space)

https://github.com/madebyollin/taesd
"""
import os
import torch
import torch.nn as nn

from modules import devices, paths_internal, shared

sd_vae_taesd_models = {}


def conv(n_in, n_out, **kwargs):
    return nn.Conv2d(n_in, n_out, 3, padding=1, **kwargs)


class Clamp(nn.Module):
    @staticmethod
    def forward(x):
        return torch.tanh(x / 3) * 3


class Block(nn.Module):
    def __init__(self, n_in, n_out):
        super().__init__()
        self.conv = nn.Sequential(conv(n_in, n_out), nn.ReLU(), conv(n_out, n_out), nn.ReLU(), conv(n_out, n_out))
        self.skip = nn.Conv2d(n_in, n_out, 1, bias=False) if n_in != n_out else nn.Identity()
        self.fuse = nn.ReLU()

    def forward(self, x):
        return self.fuse(self.conv(x) + self.skip(x))


def decoder(latent_channels=4):
    return nn.Sequential(
        Clamp(), conv(latent_channels, 64), nn.ReLU(),
        Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
        Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
        Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),
        Block(64, 64), conv(64, 3),
    )


def encoder(latent_channels=4):
    return nn.Sequential(
        conv(3, 64), Block(64, 64),
        conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
        conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
        conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),
        conv(64, latent_channels),
    )


class TAESDDecoder(nn.Module):
    latent_magnitude = 3
    latent_shift = 0.5

    def __init__(self, decoder_path="taesd_decoder.pth", latent_channels=None):
        """Initialize pretrained TAESD on the given device from the given checkpoints."""
        super().__init__()

        if latent_channels is None:
            latent_channels = 16 if "taesd3" in str(decoder_path) else 4

        self.decoder = decoder(latent_channels)
        self.decoder.load_state_dict(
            torch.load(decoder_path, map_location='cpu' if devices.device.type != 'cuda' else None))


class TAESDEncoder(nn.Module):
    latent_magnitude = 3
    latent_shift = 0.5

    def __init__(self, encoder_path="taesd_encoder.pth", latent_channels=None):
        """Initialize pretrained TAESD on the given device from the given checkpoints."""
        super().__init__()

        if latent_channels is None:
            latent_channels = 16 if "taesd3" in str(encoder_path) else 4

        self.encoder = encoder(latent_channels)
        self.encoder.load_state_dict(
            torch.load(encoder_path, map_location='cpu' if devices.device.type != 'cuda' else None))


def download_model(model_path, model_url):
    if not os.path.exists(model_path):
        os.makedirs(os.path.dirname(model_path), exist_ok=True)

        print(f'Downloading TAESD model to: {model_path}')
        torch.hub.download_url_to_file(model_url, model_path)


def decoder_model():
    if shared.sd_model.is_sd3:
        model_name = "taesd3_decoder.pth"
    elif shared.sd_model.is_sdxl:
        model_name = "taesdxl_decoder.pth"
    else:
        model_name = "taesd_decoder.pth"

    loaded_model = sd_vae_taesd_models.get(model_name)

    if loaded_model is None:
        model_path = os.path.join(paths_internal.models_path, "VAE-taesd", model_name)
        download_model(model_path, 'https://github.com/madebyollin/taesd/raw/main/' + model_name)

        if os.path.exists(model_path):
            loaded_model = TAESDDecoder(model_path)
            loaded_model.eval()
            loaded_model.to(devices.device, devices.dtype)
            sd_vae_taesd_models[model_name] = loaded_model
        else:
            raise FileNotFoundError('TAESD model not found')

    return loaded_model.decoder


def encoder_model():
    if shared.sd_model.is_sd3:
        model_name = "taesd3_encoder.pth"
    elif shared.sd_model.is_sdxl:
        model_name = "taesdxl_encoder.pth"
    else:
        model_name = "taesd_encoder.pth"

    loaded_model = sd_vae_taesd_models.get(model_name)

    if loaded_model is None:
        model_path = os.path.join(paths_internal.models_path, "VAE-taesd", model_name)
        download_model(model_path, 'https://github.com/madebyollin/taesd/raw/main/' + model_name)

        if os.path.exists(model_path):
            loaded_model = TAESDEncoder(model_path)
            loaded_model.eval()
            loaded_model.to(devices.device, devices.dtype)
            sd_vae_taesd_models[model_name] = loaded_model
        else:
            raise FileNotFoundError('TAESD model not found')

    return loaded_model.encoder
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`"""`
			`Tiny AutoEncoder for Stable Diffusion`
			`(DNN for encoding / decoding SD's latent space)`

			`https://github.com/madebyollin/taesd`
			`"""`
			`import os`
			`import torch`
			`import torch.nn as nn`

add XL support for live previews: approx and TAESD 2023-07-13 08:24:54 -06:00			`from modules import devices, paths_internal, shared`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00
add XL support for live previews: approx and TAESD 2023-07-13 08:24:54 -06:00			`sd_vae_taesd_models = {}`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00

			`def conv(n_in, n_out, **kwargs):`
			`return nn.Conv2d(n_in, n_out, 3, padding=1, **kwargs)`


			`class Clamp(nn.Module):`
			`@staticmethod`
			`def forward(x):`
			`return torch.tanh(x / 3) * 3`


			`class Block(nn.Module):`
			`def __init__(self, n_in, n_out):`
			`super().__init__()`
			`self.conv = nn.Sequential(conv(n_in, n_out), nn.ReLU(), conv(n_out, n_out), nn.ReLU(), conv(n_out, n_out))`
			`self.skip = nn.Conv2d(n_in, n_out, 1, bias=False) if n_in != n_out else nn.Identity()`
			`self.fuse = nn.ReLU()`

			`def forward(self, x):`
			`return self.fuse(self.conv(x) + self.skip(x))`


initial SD3 support 2024-06-15 23:04:31 -06:00			`def decoder(latent_channels=4):`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`return nn.Sequential(`
initial SD3 support 2024-06-15 23:04:31 -06:00			`Clamp(), conv(latent_channels, 64), nn.ReLU(),`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),`
			`Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),`
			`Block(64, 64), Block(64, 64), Block(64, 64), nn.Upsample(scale_factor=2), conv(64, 64, bias=False),`
			`Block(64, 64), conv(64, 3),`
			`)`


initial SD3 support 2024-06-15 23:04:31 -06:00			`def encoder(latent_channels=4):`
add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`return nn.Sequential(`
			`conv(3, 64), Block(64, 64),`
			`conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),`
			`conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),`
			`conv(64, 64, stride=2, bias=False), Block(64, 64), Block(64, 64), Block(64, 64),`
initial SD3 support 2024-06-15 23:04:31 -06:00			`conv(64, latent_channels),`
add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`)`


			`class TAESDDecoder(nn.Module):`
TAESD fix 2023-05-17 03:39:07 -06:00			`latent_magnitude = 3`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`latent_shift = 0.5`

initial SD3 support 2024-06-15 23:04:31 -06:00			`def __init__(self, decoder_path="taesd_decoder.pth", latent_channels=None):`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`"""Initialize pretrained TAESD on the given device from the given checkpoints."""`
			`super().__init__()`
initial SD3 support 2024-06-15 23:04:31 -06:00
			`if latent_channels is None:`
			`latent_channels = 16 if "taesd3" in str(decoder_path) else 4`

			`self.decoder = decoder(latent_channels)`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`self.decoder.load_state_dict(`
			`torch.load(decoder_path, map_location='cpu' if devices.device.type != 'cuda' else None))`

add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00
			`class TAESDEncoder(nn.Module):`
			`latent_magnitude = 3`
			`latent_shift = 0.5`

initial SD3 support 2024-06-15 23:04:31 -06:00			`def __init__(self, encoder_path="taesd_encoder.pth", latent_channels=None):`
add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`"""Initialize pretrained TAESD on the given device from the given checkpoints."""`
			`super().__init__()`
initial SD3 support 2024-06-15 23:04:31 -06:00
			`if latent_channels is None:`
			`latent_channels = 16 if "taesd3" in str(encoder_path) else 4`

			`self.encoder = encoder(latent_channels)`
add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`self.encoder.load_state_dict(`
			`torch.load(encoder_path, map_location='cpu' if devices.device.type != 'cuda' else None))`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00

add XL support for live previews: approx and TAESD 2023-07-13 08:24:54 -06:00			`def download_model(model_path, model_url):`
return live preview defaults to how they were only download TAESD model when it's needed return calculations in single_sample_to_image to just if/elif/elif blocks keep taesd model in its own directory 2023-05-17 00:24:01 -06:00			`if not os.path.exists(model_path):`
			`os.makedirs(os.path.dirname(model_path), exist_ok=True)`

add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`print(f'Downloading TAESD model to: {model_path}')`
return live preview defaults to how they were only download TAESD model when it's needed return calculations in single_sample_to_image to just if/elif/elif blocks keep taesd model in its own directory 2023-05-17 00:24:01 -06:00			`torch.hub.download_url_to_file(model_url, model_path)`


add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`def decoder_model():`
initial SD3 support 2024-06-15 23:04:31 -06:00			`if shared.sd_model.is_sd3:`
			`model_name = "taesd3_decoder.pth"`
			`elif shared.sd_model.is_sdxl:`
			`model_name = "taesdxl_decoder.pth"`
			`else:`
			`model_name = "taesd_decoder.pth"`

add XL support for live previews: approx and TAESD 2023-07-13 08:24:54 -06:00			`loaded_model = sd_vae_taesd_models.get(model_name)`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00
add XL support for live previews: approx and TAESD 2023-07-13 08:24:54 -06:00			`if loaded_model is None:`
			`model_path = os.path.join(paths_internal.models_path, "VAE-taesd", model_name)`
			`download_model(model_path, 'https://github.com/madebyollin/taesd/raw/main/' + model_name)`
return live preview defaults to how they were only download TAESD model when it's needed return calculations in single_sample_to_image to just if/elif/elif blocks keep taesd model in its own directory 2023-05-17 00:24:01 -06:00
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`if os.path.exists(model_path):`
add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`loaded_model = TAESDDecoder(model_path)`
add XL support for live previews: approx and TAESD 2023-07-13 08:24:54 -06:00			`loaded_model.eval()`
			`loaded_model.to(devices.device, devices.dtype)`
			`sd_vae_taesd_models[model_name] = loaded_model`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00			`else:`
return live preview defaults to how they were only download TAESD model when it's needed return calculations in single_sample_to_image to just if/elif/elif blocks keep taesd model in its own directory 2023-05-17 00:24:01 -06:00			`raise FileNotFoundError('TAESD model not found')`
Add Tiny AE live preview 2023-05-13 22:42:44 -06:00
add XL support for live previews: approx and TAESD 2023-07-13 08:24:54 -06:00			`return loaded_model.decoder`
add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00

			`def encoder_model():`
initial SD3 support 2024-06-15 23:04:31 -06:00			`if shared.sd_model.is_sd3:`
			`model_name = "taesd3_encoder.pth"`
			`elif shared.sd_model.is_sdxl:`
			`model_name = "taesdxl_encoder.pth"`
			`else:`
			`model_name = "taesd_encoder.pth"`

add TAESD for i2i and t2i 2023-08-03 23:38:52 -06:00			`loaded_model = sd_vae_taesd_models.get(model_name)`

			`if loaded_model is None:`
			`model_path = os.path.join(paths_internal.models_path, "VAE-taesd", model_name)`
			`download_model(model_path, 'https://github.com/madebyollin/taesd/raw/main/' + model_name)`

			`if os.path.exists(model_path):`
			`loaded_model = TAESDEncoder(model_path)`
			`loaded_model.eval()`
			`loaded_model.to(devices.device, devices.dtype)`
			`sd_vae_taesd_models[model_name] = loaded_model`
			`else:`
			`raise FileNotFoundError('TAESD model not found')`

			`return loaded_model.encoder`