Spaces:

wmpscc
/

StyleGene

Runtime error

App Files Files Community

wmpscc commited on Jun 4, 2023

Commit

7d1312d

1 Parent(s): 6aa2a6e

add

Browse files

Files changed (39) hide show

app.py +56 -0
configs.py +9 -0
data/fairface_gender_angle.csv +0 -0
img/demo.png +0 -0
img/pic_top.jpg +0 -0
models/__init__.py +0 -0
models/__pycache__/__init__.cpython-39.pyc +0 -0
models/encoders/__init__.py +0 -0
models/encoders/helpers.py +140 -0
models/encoders/model_irse.py +84 -0
models/encoders/psp_encoders.py +236 -0
models/stylegan2/__init__.py +0 -0
models/stylegan2/__pycache__/__init__.cpython-39.pyc +0 -0
models/stylegan2/__pycache__/model.cpython-39.pyc +0 -0
models/stylegan2/model.py +674 -0
models/stylegan2/op/__init__.py +2 -0
models/stylegan2/op/__pycache__/__init__.cpython-39.pyc +0 -0
models/stylegan2/op/__pycache__/fused_act.cpython-39.pyc +0 -0
models/stylegan2/op/fused_act.py +86 -0
models/stylegan2/op/fused_bias_act.cpp +21 -0
models/stylegan2/op/fused_bias_act_kernel.cu +99 -0
models/stylegan2/op/upfirdn2d.cpp +23 -0
models/stylegan2/op/upfirdn2d.py +186 -0
models/stylegan2/op/upfirdn2d_kernel.cu +272 -0
models/stylegene/__init__.py +0 -0
models/stylegene/__pycache__/__init__.cpython-39.pyc +0 -0
models/stylegene/__pycache__/api.cpython-39.pyc +0 -0
models/stylegene/api.py +94 -0
models/stylegene/data_util.py +36 -0
models/stylegene/fair_face_model.py +61 -0
models/stylegene/gene_crossover_mutation.py +64 -0
models/stylegene/gene_pool.py +42 -0
models/stylegene/model.py +209 -0
models/stylegene/util.py +30 -0
preprocess/__init__.py +0 -0
preprocess/align_images.py +32 -0
preprocess/face_alignment.py +87 -0
preprocess/landmarks_detector.py +24 -0
requirements.txt +19 -0

app.py ADDED Viewed

	@@ -0,0 +1,56 @@

+import gradio as gr
+#from models.stylegene.api import synthesize_descendant
+description = """<p style="text-align: center; font-weight: bold;">
+        <span style="font-size: 28px">StyleGene: Crossover and Mutation of Region-Level Facial Genes for Kinship Face Synthesis</span>
+        <br>
+        <span style="font-size: 18px" id="paper-info">
+            [<a href="https://wmpscc.github.io/stylegene/" target="_blank">Project Page</a>]
+            [<a href="https://openaccess.thecvf.com/content/CVPR2023/papers/Li_StyleGene_Crossover_and_Mutation_of_Region-Level_Facial_Genes_for_Kinship_CVPR_2023_paper.pdf" target="_blank">Paper</a>]
+            [<a href="https://github.com/CVI-SZU/StyleGene" target="_blank">GitHub</a>]
+        </span>
+        <br>
+        <a> Tips: One picture should have only one face.</a>
+    </p>"""
+block = gr.Blocks()
+with block:
+    gr.HTML(description)
+    with gr.Row():
+        with gr.Column():
+            gr.Markdown("### Upload photos of father and mother")
+            with gr.Row():
+                img1 = gr.Image(label="Father")
+                img2 = gr.Image(label="Mother")
+            gr.Markdown("### Select the child's age and gender")
+            with gr.Row():
+                age = gr.Dropdown(label="Age",
+                                  choices=["0-2", "3-9", "10-19", "20-29", "30-39",
+                                           "40-49", "50-59", "60-69", "70+"], value="3-9")
+                gender = gr.Dropdown(label="Gender", choices=["male", "female"], value="female")
+            gr.Markdown("### Adjust your child's resemblance to parents")
+            bar1 = gr.Slider(label="gamma", minimum=0, maximum=1, value=0.47)
+            bar2 = gr.Slider(label="eta", minimum=0, maximum=1, value=0.4)
+            bt_run = gr.Button("Run")
+            gr.Markdown("""## Disclaimer
+                            This method is intended for academic research purposes only and is strictly prohibited for commercial use.
+                            Users are required to comply with all local laws and regulations when using this method.""")
+        with gr.Column():
+            gr.Markdown("### Results")
+            img3 = gr.Image(label="Generated child")
+            with gr.Row():
+                img1_align = gr.Image(label="Father")
+                img2_align = gr.Image(label="Mother")
+    def run(father, mother, gamma, eta, age, gender):
+        attributes = {'age': age, 'gender': gender, 'gamma': float(gamma), 'eta': float(eta)}
+        img_F, img_M, img_C = synthesize_descendant(father, mother, attributes)
+        return img_F, img_M, img_C
+    bt_run.click(run, [img1, img2, bar1, bar2, age, gender], [img1_align, img2_align, img3])
+block.launch(show_error=True)

configs.py ADDED Viewed

	@@ -0,0 +1,9 @@

+path_ckpt_landmark68 = "checkpoints/shape_predictor_68_face_landmarks.dat.bz2"
+path_ckpt_e4e = "/home/cvi_demo/PythonProject/StyleGene/checkpoints/e4e_ffhq_encode.pt"
+path_ckpt_stylegan2 = '/home/cvi_demo/PythonProject/StyleGene/checkpoints/stylegan2-ffhq-config-f.pt'
+path_ckpt_stylegene = "/home/cvi_demo/PythonProject/StyleGene/checkpoints/stylegene_N18.ckpt"
+path_ckpt_fairface = '/home/cvi_demo/PythonProject/StyleGene/checkpoints/res34_fair_align_multi_7_20190809.pt'
+path_ckpt_genepool = "/home/cvi_demo/PythonProject/StyleGene/checkpoints/geneFactorPool.pkl"
+path_csv_ffhq_attritube = 'data/fairface_gender_angle.csv'
+path_dataset_ffhq = None

data/fairface_gender_angle.csv ADDED Viewed

The diff for this file is too large to render. See raw diff

img/demo.png ADDED Viewed

img/pic_top.jpg ADDED Viewed

models/__init__.py ADDED Viewed

File without changes

models/__pycache__/__init__.cpython-39.pyc ADDED Viewed

Binary file (135 Bytes). View file

models/encoders/__init__.py ADDED Viewed

File without changes

models/encoders/helpers.py ADDED Viewed

	@@ -0,0 +1,140 @@

+from collections import namedtuple
+import torch
+import torch.nn.functional as F
+from torch.nn import Conv2d, BatchNorm2d, PReLU, ReLU, Sigmoid, MaxPool2d, AdaptiveAvgPool2d, Sequential, Module
+"""
+ArcFace implementation from [TreB1eN](https://github.com/TreB1eN/InsightFace_Pytorch)
+"""
+class Flatten(Module):
+    def forward(self, input):
+        return input.view(input.size(0), -1)
+def l2_norm(input, axis=1):
+    norm = torch.norm(input, 2, axis, True)
+    output = torch.div(input, norm)
+    return output
+class Bottleneck(namedtuple('Block', ['in_channel', 'depth', 'stride'])):
+    """ A named tuple describing a ResNet block. """
+def get_block(in_channel, depth, num_units, stride=2):
+    return [Bottleneck(in_channel, depth, stride)] + [Bottleneck(depth, depth, 1) for i in range(num_units - 1)]
+def get_blocks(num_layers):
+    if num_layers == 50:
+        blocks = [
+            get_block(in_channel=64, depth=64, num_units=3),
+            get_block(in_channel=64, depth=128, num_units=4),
+            get_block(in_channel=128, depth=256, num_units=14),
+            get_block(in_channel=256, depth=512, num_units=3)
+        ]
+    elif num_layers == 100:
+        blocks = [
+            get_block(in_channel=64, depth=64, num_units=3),
+            get_block(in_channel=64, depth=128, num_units=13),
+            get_block(in_channel=128, depth=256, num_units=30),
+            get_block(in_channel=256, depth=512, num_units=3)
+        ]
+    elif num_layers == 152:
+        blocks = [
+            get_block(in_channel=64, depth=64, num_units=3),
+            get_block(in_channel=64, depth=128, num_units=8),
+            get_block(in_channel=128, depth=256, num_units=36),
+            get_block(in_channel=256, depth=512, num_units=3)
+        ]
+    else:
+        raise ValueError("Invalid number of layers: {}. Must be one of [50, 100, 152]".format(num_layers))
+    return blocks
+class SEModule(Module):
+    def __init__(self, channels, reduction):
+        super(SEModule, self).__init__()
+        self.avg_pool = AdaptiveAvgPool2d(1)
+        self.fc1 = Conv2d(channels, channels // reduction, kernel_size=1, padding=0, bias=False)
+        self.relu = ReLU(inplace=True)
+        self.fc2 = Conv2d(channels // reduction, channels, kernel_size=1, padding=0, bias=False)
+        self.sigmoid = Sigmoid()
+    def forward(self, x):
+        module_input = x
+        x = self.avg_pool(x)
+        x = self.fc1(x)
+        x = self.relu(x)
+        x = self.fc2(x)
+        x = self.sigmoid(x)
+        return module_input * x
+class bottleneck_IR(Module):
+    def __init__(self, in_channel, depth, stride):
+        super(bottleneck_IR, self).__init__()
+        if in_channel == depth:
+            self.shortcut_layer = MaxPool2d(1, stride)
+        else:
+            self.shortcut_layer = Sequential(
+                Conv2d(in_channel, depth, (1, 1), stride, bias=False),
+                BatchNorm2d(depth)
+            )
+        self.res_layer = Sequential(
+            BatchNorm2d(in_channel),
+            Conv2d(in_channel, depth, (3, 3), (1, 1), 1, bias=False), PReLU(depth),
+            Conv2d(depth, depth, (3, 3), stride, 1, bias=False), BatchNorm2d(depth)
+        )
+    def forward(self, x):
+        shortcut = self.shortcut_layer(x)
+        res = self.res_layer(x)
+        return res + shortcut
+class bottleneck_IR_SE(Module):
+    def __init__(self, in_channel, depth, stride):
+        super(bottleneck_IR_SE, self).__init__()
+        if in_channel == depth:
+            self.shortcut_layer = MaxPool2d(1, stride)
+        else:
+            self.shortcut_layer = Sequential(
+                Conv2d(in_channel, depth, (1, 1), stride, bias=False),
+                BatchNorm2d(depth)
+            )
+        self.res_layer = Sequential(
+            BatchNorm2d(in_channel),
+            Conv2d(in_channel, depth, (3, 3), (1, 1), 1, bias=False),
+            PReLU(depth),
+            Conv2d(depth, depth, (3, 3), stride, 1, bias=False),
+            BatchNorm2d(depth),
+            SEModule(depth, 16)
+        )
+    def forward(self, x):
+        shortcut = self.shortcut_layer(x)
+        res = self.res_layer(x)
+        return res + shortcut
+def _upsample_add(x, y):
+    """Upsample and add two feature maps.
+    Args:
+      x: (Variable) top feature map to be upsampled.
+      y: (Variable) lateral feature map.
+    Returns:
+      (Variable) added feature map.
+    Note in PyTorch, when input size is odd, the upsampled feature map
+    with `F.upsample(..., scale_factor=2, mode='nearest')`
+    maybe not equal to the lateral feature map size.
+    e.g.
+    original input size: [N,_,15,15] ->
+    conv2d feature map size: [N,_,8,8] ->
+    upsampled feature map size: [N,_,16,16]
+    So we choose bilinear upsample which supports arbitrary output sizes.
+    """
+    _, _, H, W = y.size()
+    return F.interpolate(x, size=(H, W), mode='bilinear', align_corners=True) + y

models/encoders/model_irse.py ADDED Viewed

	@@ -0,0 +1,84 @@

+from torch.nn import Linear, Conv2d, BatchNorm1d, BatchNorm2d, PReLU, Dropout, Sequential, Module
+from models.encoders.helpers import get_blocks, Flatten, bottleneck_IR, bottleneck_IR_SE, l2_norm
+"""
+Modified Backbone implementation from [TreB1eN](https://github.com/TreB1eN/InsightFace_Pytorch)
+"""
+class Backbone(Module):
+    def __init__(self, input_size, num_layers, mode='ir', drop_ratio=0.4, affine=True):
+        super(Backbone, self).__init__()
+        assert input_size in [112, 224], "input_size should be 112 or 224"
+        assert num_layers in [50, 100, 152], "num_layers should be 50, 100 or 152"
+        assert mode in ['ir', 'ir_se'], "mode should be ir or ir_se"
+        blocks = get_blocks(num_layers)
+        if mode == 'ir':
+            unit_module = bottleneck_IR
+        elif mode == 'ir_se':
+            unit_module = bottleneck_IR_SE
+        self.input_layer = Sequential(Conv2d(3, 64, (3, 3), 1, 1, bias=False),
+                                      BatchNorm2d(64),
+                                      PReLU(64))
+        if input_size == 112:
+            self.output_layer = Sequential(BatchNorm2d(512),
+                                           Dropout(drop_ratio),
+                                           Flatten(),
+                                           Linear(512 * 7 * 7, 512),
+                                           BatchNorm1d(512, affine=affine))
+        else:
+            self.output_layer = Sequential(BatchNorm2d(512),
+                                           Dropout(drop_ratio),
+                                           Flatten(),
+                                           Linear(512 * 14 * 14, 512),
+                                           BatchNorm1d(512, affine=affine))
+        modules = []
+        for block in blocks:
+            for bottleneck in block:
+                modules.append(unit_module(bottleneck.in_channel,
+                                           bottleneck.depth,
+                                           bottleneck.stride))
+        self.body = Sequential(*modules)
+    def forward(self, x):
+        x = self.input_layer(x)
+        x = self.body(x)
+        x = self.output_layer(x)
+        return l2_norm(x)
+def IR_50(input_size):
+    """Constructs a ir-50 model."""
+    model = Backbone(input_size, num_layers=50, mode='ir', drop_ratio=0.4, affine=False)
+    return model
+def IR_101(input_size):
+    """Constructs a ir-101 model."""
+    model = Backbone(input_size, num_layers=100, mode='ir', drop_ratio=0.4, affine=False)
+    return model
+def IR_152(input_size):
+    """Constructs a ir-152 model."""
+    model = Backbone(input_size, num_layers=152, mode='ir', drop_ratio=0.4, affine=False)
+    return model
+def IR_SE_50(input_size):
+    """Constructs a ir_se-50 model."""
+    model = Backbone(input_size, num_layers=50, mode='ir_se', drop_ratio=0.4, affine=False)
+    return model
+def IR_SE_101(input_size):
+    """Constructs a ir_se-101 model."""
+    model = Backbone(input_size, num_layers=100, mode='ir_se', drop_ratio=0.4, affine=False)
+    return model
+def IR_SE_152(input_size):
+    """Constructs a ir_se-152 model."""
+    model = Backbone(input_size, num_layers=152, mode='ir_se', drop_ratio=0.4, affine=False)
+    return model

models/encoders/psp_encoders.py ADDED Viewed

	@@ -0,0 +1,236 @@

+from enum import Enum
+import math
+import numpy as np
+import torch
+from torch import nn
+from torch.nn import Conv2d, BatchNorm2d, PReLU, Sequential, Module
+from models.encoders.helpers import get_blocks, bottleneck_IR, bottleneck_IR_SE, _upsample_add
+from models.stylegan2.model import EqualLinear
+# Adapted from https://github.com/omertov/encoder4editing
+class ProgressiveStage(Enum):
+    WTraining = 0
+    Delta1Training = 1
+    Delta2Training = 2
+    Delta3Training = 3
+    Delta4Training = 4
+    Delta5Training = 5
+    Delta6Training = 6
+    Delta7Training = 7
+    Delta8Training = 8
+    Delta9Training = 9
+    Delta10Training = 10
+    Delta11Training = 11
+    Delta12Training = 12
+    Delta13Training = 13
+    Delta14Training = 14
+    Delta15Training = 15
+    Delta16Training = 16
+    Delta17Training = 17
+    Inference = 18
+class GradualStyleBlock(Module):
+    def __init__(self, in_c, out_c, spatial):
+        super(GradualStyleBlock, self).__init__()
+        self.out_c = out_c
+        self.spatial = spatial
+        num_pools = int(np.log2(spatial))
+        modules = []
+        modules += [Conv2d(in_c, out_c, kernel_size=3, stride=2, padding=1),
+                    nn.LeakyReLU()]
+        for i in range(num_pools - 1):
+            modules += [
+                Conv2d(out_c, out_c, kernel_size=3, stride=2, padding=1),
+                nn.LeakyReLU()
+            ]
+        self.convs = nn.Sequential(*modules)
+        self.linear = EqualLinear(out_c, out_c, lr_mul=1)
+    def forward(self, x):
+        x = self.convs(x)
+        x = x.view(-1, self.out_c)
+        x = self.linear(x)
+        return x
+class GradualStyleEncoder(Module):
+    def __init__(self, num_layers, mode='ir', opts=None):
+        super(GradualStyleEncoder, self).__init__()
+        assert num_layers in [50, 100, 152], 'num_layers should be 50,100, or 152'
+        assert mode in ['ir', 'ir_se'], 'mode should be ir or ir_se'
+        blocks = get_blocks(num_layers)
+        if mode == 'ir':
+            unit_module = bottleneck_IR
+        elif mode == 'ir_se':
+            unit_module = bottleneck_IR_SE
+        self.input_layer = Sequential(Conv2d(3, 64, (3, 3), 1, 1, bias=False),
+                                      BatchNorm2d(64),
+                                      PReLU(64))
+        modules = []
+        for block in blocks:
+            for bottleneck in block:
+                modules.append(unit_module(bottleneck.in_channel,
+                                           bottleneck.depth,
+                                           bottleneck.stride))
+        self.body = Sequential(*modules)
+        self.styles = nn.ModuleList()
+        log_size = int(math.log(opts.stylegan_size, 2))
+        self.style_count = 2 * log_size - 2
+        self.coarse_ind = 3
+        self.middle_ind = 7
+        for i in range(self.style_count):
+            if i < self.coarse_ind:
+                style = GradualStyleBlock(512, 512, 16)
+            elif i < self.middle_ind:
+                style = GradualStyleBlock(512, 512, 32)
+            else:
+                style = GradualStyleBlock(512, 512, 64)
+            self.styles.append(style)
+        self.latlayer1 = nn.Conv2d(256, 512, kernel_size=1, stride=1, padding=0)
+        self.latlayer2 = nn.Conv2d(128, 512, kernel_size=1, stride=1, padding=0)
+    def forward(self, x):
+        x = self.input_layer(x)
+        latents = []
+        modulelist = list(self.body._modules.values())
+        for i, l in enumerate(modulelist):
+            x = l(x)
+            if i == 6:
+                c1 = x
+            elif i == 20:
+                c2 = x
+            elif i == 23:
+                c3 = x
+        for j in range(self.coarse_ind):
+            latents.append(self.styles[j](c3))
+        p2 = _upsample_add(c3, self.latlayer1(c2))
+        for j in range(self.coarse_ind, self.middle_ind):
+            latents.append(self.styles[j](p2))
+        p1 = _upsample_add(p2, self.latlayer2(c1))
+        for j in range(self.middle_ind, self.style_count):
+            latents.append(self.styles[j](p1))
+        out = torch.stack(latents, dim=1)
+        return out
+class Encoder4Editing(Module):
+    def __init__(self, num_layers, mode='ir', stylegan_size=1024):
+        super(Encoder4Editing, self).__init__()
+        assert num_layers in [50, 100, 152], 'num_layers should be 50,100, or 152'
+        assert mode in ['ir', 'ir_se'], 'mode should be ir or ir_se'
+        blocks = get_blocks(num_layers)
+        if mode == 'ir':
+            unit_module = bottleneck_IR
+        elif mode == 'ir_se':
+            unit_module = bottleneck_IR_SE
+        self.input_layer = Sequential(Conv2d(3, 64, (3, 3), 1, 1, bias=False),
+                                      BatchNorm2d(64),
+                                      PReLU(64))
+        modules = []
+        for block in blocks:
+            for bottleneck in block:
+                modules.append(unit_module(bottleneck.in_channel,
+                                           bottleneck.depth,
+                                           bottleneck.stride))
+        self.body = Sequential(*modules)
+        self.styles = nn.ModuleList()
+        log_size = int(math.log(stylegan_size, 2))
+        self.style_count = 2 * log_size - 2
+        self.coarse_ind = 3
+        self.middle_ind = 7
+        for i in range(self.style_count):
+            if i < self.coarse_ind:
+                style = GradualStyleBlock(512, 512, 16)
+            elif i < self.middle_ind:
+                style = GradualStyleBlock(512, 512, 32)
+            else:
+                style = GradualStyleBlock(512, 512, 64)
+            self.styles.append(style)
+        self.latlayer1 = nn.Conv2d(256, 512, kernel_size=1, stride=1, padding=0)
+        self.latlayer2 = nn.Conv2d(128, 512, kernel_size=1, stride=1, padding=0)
+        self.progressive_stage = ProgressiveStage.Inference
+    def get_deltas_starting_dimensions(self):
+        ''' Get a list of the initial dimension of every delta from which it is applied '''
+        return list(range(self.style_count))  # Each dimension has a delta applied to it
+    def set_progressive_stage(self, new_stage: ProgressiveStage):
+        self.progressive_stage = new_stage
+        print('Changed progressive stage to: ', new_stage)
+    def forward(self, x):
+        x = self.input_layer(x)
+        modulelist = list(self.body._modules.values())
+        for i, l in enumerate(modulelist):
+            x = l(x)
+            if i == 6:
+                c1 = x
+            elif i == 20:
+                c2 = x
+            elif i == 23:
+                c3 = x
+        # Infer main W and duplicate it
+        w0 = self.styles[0](c3)
+        w = w0.repeat(self.style_count, 1, 1).permute(1, 0, 2)
+        stage = self.progressive_stage.value
+        features = c3
+        for i in range(1, min(stage + 1, self.style_count)):  # Infer additional deltas
+            if i == self.coarse_ind:
+                p2 = _upsample_add(c3, self.latlayer1(c2))  # FPN's middle features
+                features = p2
+            elif i == self.middle_ind:
+                p1 = _upsample_add(p2, self.latlayer2(c1))  # FPN's fine features
+                features = p1
+            delta_i = self.styles[i](features)
+            w[:, i] += delta_i
+        return w
+class BackboneEncoderUsingLastLayerIntoW(Module):
+    def __init__(self, num_layers, mode='ir', opts=None):
+        super(BackboneEncoderUsingLastLayerIntoW, self).__init__()
+        print('Using BackboneEncoderUsingLastLayerIntoW')
+        assert num_layers in [50, 100, 152], 'num_layers should be 50,100, or 152'
+        assert mode in ['ir', 'ir_se'], 'mode should be ir or ir_se'
+        blocks = get_blocks(num_layers)
+        if mode == 'ir':
+            unit_module = bottleneck_IR
+        elif mode == 'ir_se':
+            unit_module = bottleneck_IR_SE
+        self.input_layer = Sequential(Conv2d(3, 64, (3, 3), 1, 1, bias=False),
+                                      BatchNorm2d(64),
+                                      PReLU(64))
+        self.output_pool = torch.nn.AdaptiveAvgPool2d((1, 1))
+        self.linear = EqualLinear(512, 512, lr_mul=1)
+        modules = []
+        for block in blocks:
+            for bottleneck in block:
+                modules.append(unit_module(bottleneck.in_channel,
+                                           bottleneck.depth,
+                                           bottleneck.stride))
+        self.body = Sequential(*modules)
+        log_size = int(math.log(opts.stylegan_size, 2))
+        self.style_count = 2 * log_size - 2
+    def forward(self, x):
+        x = self.input_layer(x)
+        x = self.body(x)
+        x = self.output_pool(x)
+        x = x.view(-1, 512)
+        x = self.linear(x)
+        return x.repeat(self.style_count, 1, 1).permute(1, 0, 2)

models/stylegan2/__init__.py ADDED Viewed

File without changes

models/stylegan2/__pycache__/__init__.cpython-39.pyc ADDED Viewed

Binary file (145 Bytes). View file

models/stylegan2/__pycache__/model.cpython-39.pyc ADDED Viewed

Binary file (15.8 kB). View file

models/stylegan2/model.py ADDED Viewed

	@@ -0,0 +1,674 @@

+import math
+import random
+import torch
+from torch import nn
+from torch.nn import functional as F
+from .op import FusedLeakyReLU, fused_leaky_relu, upfirdn2d
+# Adapted from https://github.com/rosinality/stylegan2-pytorch
+class PixelNorm(nn.Module):
+    def __init__(self):
+        super().__init__()
+    def forward(self, input):
+        return input * torch.rsqrt(torch.mean(input ** 2, dim=1, keepdim=True) + 1e-8)
+def make_kernel(k):
+    k = torch.tensor(k, dtype=torch.float32)
+    if k.ndim == 1:
+        k = k[None, :] * k[:, None]
+    k /= k.sum()
+    return k
+class Upsample(nn.Module):
+    def __init__(self, kernel, factor=2):
+        super().__init__()
+        self.factor = factor
+        kernel = make_kernel(kernel) * (factor ** 2)
+        self.register_buffer('kernel', kernel)
+        p = kernel.shape[0] - factor
+        pad0 = (p + 1) // 2 + factor - 1
+        pad1 = p // 2
+        self.pad = (pad0, pad1)
+    def forward(self, input):
+        out = upfirdn2d(input, self.kernel, up=self.factor, down=1, pad=self.pad)
+        return out
+class Downsample(nn.Module):
+    def __init__(self, kernel, factor=2):
+        super().__init__()
+        self.factor = factor
+        kernel = make_kernel(kernel)
+        self.register_buffer('kernel', kernel)
+        p = kernel.shape[0] - factor
+        pad0 = (p + 1) // 2
+        pad1 = p // 2
+        self.pad = (pad0, pad1)
+    def forward(self, input):
+        out = upfirdn2d(input, self.kernel, up=1, down=self.factor, pad=self.pad)
+        return out
+class Blur(nn.Module):
+    def __init__(self, kernel, pad, upsample_factor=1):
+        super().__init__()
+        kernel = make_kernel(kernel)
+        if upsample_factor > 1:
+            kernel = kernel * (upsample_factor ** 2)
+        self.register_buffer('kernel', kernel)
+        self.pad = pad
+    def forward(self, input):
+        out = upfirdn2d(input, self.kernel, pad=self.pad)
+        return out
+class EqualConv2d(nn.Module):
+    def __init__(
+            self, in_channel, out_channel, kernel_size, stride=1, padding=0, bias=True
+    ):
+        super().__init__()
+        self.weight = nn.Parameter(
+            torch.randn(out_channel, in_channel, kernel_size, kernel_size)
+        )
+        self.scale = 1 / math.sqrt(in_channel * kernel_size ** 2)
+        self.stride = stride
+        self.padding = padding
+        if bias:
+            self.bias = nn.Parameter(torch.zeros(out_channel))
+        else:
+            self.bias = None
+    def forward(self, input):
+        out = F.conv2d(
+            input,
+            self.weight * self.scale,
+            bias=self.bias,
+            stride=self.stride,
+            padding=self.padding,
+        )
+        return out
+    def __repr__(self):
+        return (
+            f'{self.__class__.__name__}({self.weight.shape[1]}, {self.weight.shape[0]},'
+            f' {self.weight.shape[2]}, stride={self.stride}, padding={self.padding})'
+        )
+class EqualLinear(nn.Module):
+    def __init__(
+            self, in_dim, out_dim, bias=True, bias_init=0, lr_mul=1, activation=None
+    ):
+        super().__init__()
+        self.weight = nn.Parameter(torch.randn(out_dim, in_dim).div_(lr_mul))
+        if bias:
+            self.bias = nn.Parameter(torch.zeros(out_dim).fill_(bias_init))
+        else:
+            self.bias = None
+        self.activation = activation
+        self.scale = (1 / math.sqrt(in_dim)) * lr_mul
+        self.lr_mul = lr_mul
+    def forward(self, input):
+        if self.activation:
+            out = F.linear(input, self.weight * self.scale)
+            out = fused_leaky_relu(out, self.bias * self.lr_mul)
+        else:
+            out = F.linear(
+                input, self.weight * self.scale, bias=self.bias * self.lr_mul
+            )
+        return out
+    def __repr__(self):
+        return (
+            f'{self.__class__.__name__}({self.weight.shape[1]}, {self.weight.shape[0]})'
+        )
+class ScaledLeakyReLU(nn.Module):
+    def __init__(self, negative_slope=0.2):
+        super().__init__()
+        self.negative_slope = negative_slope
+    def forward(self, input):
+        out = F.leaky_relu(input, negative_slope=self.negative_slope)
+        return out * math.sqrt(2)
+class ModulatedConv2d(nn.Module):
+    def __init__(
+            self,
+            in_channel,
+            out_channel,
+            kernel_size,
+            style_dim,
+            demodulate=True,
+            upsample=False,
+            downsample=False,
+            blur_kernel=[1, 3, 3, 1],
+    ):
+        super().__init__()
+        self.eps = 1e-8
+        self.kernel_size = kernel_size
+        self.in_channel = in_channel
+        self.out_channel = out_channel
+        self.upsample = upsample
+        self.downsample = downsample
+        if upsample:
+            factor = 2
+            p = (len(blur_kernel) - factor) - (kernel_size - 1)
+            pad0 = (p + 1) // 2 + factor - 1
+            pad1 = p // 2 + 1
+            self.blur = Blur(blur_kernel, pad=(pad0, pad1), upsample_factor=factor)
+        if downsample:
+            factor = 2
+            p = (len(blur_kernel) - factor) + (kernel_size - 1)
+            pad0 = (p + 1) // 2
+            pad1 = p // 2
+            self.blur = Blur(blur_kernel, pad=(pad0, pad1))
+        fan_in = in_channel * kernel_size ** 2
+        self.scale = 1 / math.sqrt(fan_in)
+        self.padding = kernel_size // 2
+        self.weight = nn.Parameter(
+            torch.randn(1, out_channel, in_channel, kernel_size, kernel_size)
+        )
+        self.modulation = EqualLinear(style_dim, in_channel, bias_init=1)
+        self.demodulate = demodulate
+    def __repr__(self):
+        return (
+            f'{self.__class__.__name__}({self.in_channel}, {self.out_channel}, {self.kernel_size}, '
+            f'upsample={self.upsample}, downsample={self.downsample})'
+        )
+    def forward(self, input, style):
+        batch, in_channel, height, width = input.shape
+        style = self.modulation(style).view(batch, 1, in_channel, 1, 1)
+        weight = self.scale * self.weight * style
+        if self.demodulate:
+            demod = torch.rsqrt(weight.pow(2).sum([2, 3, 4]) + 1e-8)
+            weight = weight * demod.view(batch, self.out_channel, 1, 1, 1)
+        weight = weight.view(
+            batch * self.out_channel, in_channel, self.kernel_size, self.kernel_size
+        )
+        if self.upsample:
+            input = input.view(1, batch * in_channel, height, width)
+            weight = weight.view(
+                batch, self.out_channel, in_channel, self.kernel_size, self.kernel_size
+            )
+            weight = weight.transpose(1, 2).reshape(
+                batch * in_channel, self.out_channel, self.kernel_size, self.kernel_size
+            )
+            out = F.conv_transpose2d(input, weight, padding=0, stride=2, groups=batch)
+            _, _, height, width = out.shape
+            out = out.view(batch, self.out_channel, height, width)
+            out = self.blur(out)
+        elif self.downsample:
+            input = self.blur(input)
+            _, _, height, width = input.shape
+            input = input.view(1, batch * in_channel, height, width)
+            out = F.conv2d(input, weight, padding=0, stride=2, groups=batch)
+            _, _, height, width = out.shape
+            out = out.view(batch, self.out_channel, height, width)
+        else:
+            input = input.view(1, batch * in_channel, height, width)
+            out = F.conv2d(input, weight, padding=self.padding, groups=batch)
+            _, _, height, width = out.shape
+            out = out.view(batch, self.out_channel, height, width)
+        return out
+class NoiseInjection(nn.Module):
+    def __init__(self):
+        super().__init__()
+        self.weight = nn.Parameter(torch.zeros(1))
+    def forward(self, image, noise=None):
+        if noise is None:
+            batch, _, height, width = image.shape
+            noise = image.new_empty(batch, 1, height, width).normal_()
+        return image + self.weight * noise
+class ConstantInput(nn.Module):
+    def __init__(self, channel, size=4):
+        super().__init__()
+        self.input = nn.Parameter(torch.randn(1, channel, size, size))
+    def forward(self, input):
+        batch = input.shape[0]
+        out = self.input.repeat(batch, 1, 1, 1)
+        return out
+class StyledConv(nn.Module):
+    def __init__(
+            self,
+            in_channel,
+            out_channel,
+            kernel_size,
+            style_dim,
+            upsample=False,
+            blur_kernel=[1, 3, 3, 1],
+            demodulate=True,
+    ):
+        super().__init__()
+        self.conv = ModulatedConv2d(
+            in_channel,
+            out_channel,
+            kernel_size,
+            style_dim,
+            upsample=upsample,
+            blur_kernel=blur_kernel,
+            demodulate=demodulate,
+        )
+        self.noise = NoiseInjection()
+        # self.bias = nn.Parameter(torch.zeros(1, out_channel, 1, 1))
+        # self.activate = ScaledLeakyReLU(0.2)
+        self.activate = FusedLeakyReLU(out_channel)
+    def forward(self, input, style, noise=None):
+        out = self.conv(input, style)
+        out = self.noise(out, noise=noise)
+        # out = out + self.bias
+        out = self.activate(out)
+        return out
+class ToRGB(nn.Module):
+    def __init__(self, in_channel, style_dim, upsample=True, blur_kernel=[1, 3, 3, 1]):
+        super().__init__()
+        if upsample:
+            self.upsample = Upsample(blur_kernel)
+        self.conv = ModulatedConv2d(in_channel, 3, 1, style_dim, demodulate=False)
+        self.bias = nn.Parameter(torch.zeros(1, 3, 1, 1))
+    def forward(self, input, style, skip=None):
+        out = self.conv(input, style)
+        out = out + self.bias
+        if skip is not None:
+            skip = self.upsample(skip)
+            out = out + skip
+        return out
+class Generator(nn.Module):
+    def __init__(
+            self,
+            size,
+            style_dim,
+            n_mlp,
+            channel_multiplier=2,
+            blur_kernel=[1, 3, 3, 1],
+            lr_mlp=0.01,
+    ):
+        super().__init__()
+        self.size = size
+        self.style_dim = style_dim
+        layers = [PixelNorm()]
+        for i in range(n_mlp):
+            layers.append(
+                EqualLinear(
+                    style_dim, style_dim, lr_mul=lr_mlp, activation='fused_lrelu'
+                )
+            )
+        self.style = nn.Sequential(*layers)
+        self.channels = {
+            4: 512,
+            8: 512,
+            16: 512,
+            32: 512,
+            64: 256 * channel_multiplier,
+            128: 128 * channel_multiplier,
+            256: 64 * channel_multiplier,
+            512: 32 * channel_multiplier,
+            1024: 16 * channel_multiplier,
+        }
+        self.input = ConstantInput(self.channels[4])
+        self.conv1 = StyledConv(
+            self.channels[4], self.channels[4], 3, style_dim, blur_kernel=blur_kernel
+        )
+        self.to_rgb1 = ToRGB(self.channels[4], style_dim, upsample=False)
+        self.log_size = int(math.log(size, 2))
+        self.num_layers = (self.log_size - 2) * 2 + 1
+        self.convs = nn.ModuleList()
+        self.upsamples = nn.ModuleList()
+        self.to_rgbs = nn.ModuleList()
+        self.noises = nn.Module()
+        in_channel = self.channels[4]
+        for layer_idx in range(self.num_layers):
+            res = (layer_idx + 5) // 2
+            shape = [1, 1, 2 ** res, 2 ** res]
+            self.noises.register_buffer(f'noise_{layer_idx}', torch.randn(*shape))
+        for i in range(3, self.log_size + 1):
+            out_channel = self.channels[2 ** i]
+            self.convs.append(
+                StyledConv(
+                    in_channel,
+                    out_channel,
+                    3,
+                    style_dim,
+                    upsample=True,
+                    blur_kernel=blur_kernel,
+                )
+            )
+            self.convs.append(
+                StyledConv(
+                    out_channel, out_channel, 3, style_dim, blur_kernel=blur_kernel
+                )
+            )
+            self.to_rgbs.append(ToRGB(out_channel, style_dim))
+            in_channel = out_channel
+        self.n_latent = self.log_size * 2 - 2
+    def make_noise(self):
+        device = self.input.input.device
+        noises = [torch.randn(1, 1, 2 ** 2, 2 ** 2, device=device)]
+        for i in range(3, self.log_size + 1):
+            for _ in range(2):
+                noises.append(torch.randn(1, 1, 2 ** i, 2 ** i, device=device))
+        return noises
+    def mean_latent(self, n_latent):
+        latent_in = torch.randn(
+            n_latent, self.style_dim, device=self.input.input.device
+        )
+        latent = self.style(latent_in).mean(0, keepdim=True)
+        return latent
+    def get_latent(self, input):
+        return self.style(input)
+    def forward(
+            self,
+            styles,
+            return_latents=False,
+            return_features=False,
+            inject_index=None,
+            truncation=1,
+            truncation_latent=None,
+            input_is_latent=False,
+            noise=None,
+            randomize_noise=True,
+    ):
+        if not input_is_latent:
+            styles = [self.style(s) for s in styles]
+        if noise is None:
+            if randomize_noise:
+                noise = [None] * self.num_layers
+            else:
+                noise = [
+                    getattr(self.noises, f'noise_{i}') for i in range(self.num_layers)
+                ]
+        if truncation < 1:
+            style_t = []
+            for style in styles:
+                style_t.append(
+                    truncation_latent + truncation * (style - truncation_latent)
+                )
+            styles = style_t
+        if len(styles) < 2:
+            inject_index = self.n_latent
+            if styles[0].ndim < 3:
+                latent = styles[0].unsqueeze(1).repeat(1, inject_index, 1)
+            else:
+                latent = styles[0]
+        else:
+            if inject_index is None:
+                inject_index = random.randint(1, self.n_latent - 1)
+            latent = styles[0].unsqueeze(1).repeat(1, inject_index, 1)
+            latent2 = styles[1].unsqueeze(1).repeat(1, self.n_latent - inject_index, 1)
+            latent = torch.cat([latent, latent2], 1)
+        out = self.input(latent)
+        out = self.conv1(out, latent[:, 0], noise=noise[0])
+        skip = self.to_rgb1(out, latent[:, 1])
+        i = 1
+        for conv1, conv2, noise1, noise2, to_rgb in zip(
+                self.convs[::2], self.convs[1::2], noise[1::2], noise[2::2], self.to_rgbs
+        ):
+            out = conv1(out, latent[:, i], noise=noise1)
+            out = conv2(out, latent[:, i + 1], noise=noise2)
+            skip = to_rgb(out, latent[:, i + 2], skip)
+            i += 2
+        image = skip
+        if return_latents:
+            return image, latent
+        elif return_features:
+            return image, out
+        else:
+            return image, None
+class ConvLayer(nn.Sequential):
+    def __init__(
+            self,
+            in_channel,
+            out_channel,
+            kernel_size,
+            downsample=False,
+            blur_kernel=[1, 3, 3, 1],
+            bias=True,
+            activate=True,
+    ):
+        layers = []
+        if downsample:
+            factor = 2
+            p = (len(blur_kernel) - factor) + (kernel_size - 1)
+            pad0 = (p + 1) // 2
+            pad1 = p // 2
+            layers.append(Blur(blur_kernel, pad=(pad0, pad1)))
+            stride = 2
+            self.padding = 0
+        else:
+            stride = 1
+            self.padding = kernel_size // 2
+        layers.append(
+            EqualConv2d(
+                in_channel,
+                out_channel,
+                kernel_size,
+                padding=self.padding,
+                stride=stride,
+                bias=bias and not activate,
+            )
+        )
+        if activate:
+            if bias:
+                layers.append(FusedLeakyReLU(out_channel))
+            else:
+                layers.append(ScaledLeakyReLU(0.2))
+        super().__init__(*layers)
+class ResBlock(nn.Module):
+    def __init__(self, in_channel, out_channel, blur_kernel=[1, 3, 3, 1]):
+        super().__init__()
+        self.conv1 = ConvLayer(in_channel, in_channel, 3)
+        self.conv2 = ConvLayer(in_channel, out_channel, 3, downsample=True)
+        self.skip = ConvLayer(
+            in_channel, out_channel, 1, downsample=True, activate=False, bias=False
+        )
+    def forward(self, input):
+        out = self.conv1(input)
+        out = self.conv2(out)
+        skip = self.skip(input)
+        out = (out + skip) / math.sqrt(2)
+        return out
+class Discriminator(nn.Module):
+    def __init__(self, size, channel_multiplier=2, blur_kernel=[1, 3, 3, 1]):
+        super().__init__()
+        channels = {
+            4: 512,
+            8: 512,
+            16: 512,
+            32: 512,
+            64: 256 * channel_multiplier,
+            128: 128 * channel_multiplier,
+            256: 64 * channel_multiplier,
+            512: 32 * channel_multiplier,
+            1024: 16 * channel_multiplier,
+        }
+        convs = [ConvLayer(3, channels[size], 1)]
+        log_size = int(math.log(size, 2))
+        in_channel = channels[size]
+        for i in range(log_size, 2, -1):
+            out_channel = channels[2 ** (i - 1)]
+            convs.append(ResBlock(in_channel, out_channel, blur_kernel))
+            in_channel = out_channel
+        self.convs = nn.Sequential(*convs)
+        self.stddev_group = 4
+        self.stddev_feat = 1
+        self.final_conv = ConvLayer(in_channel + 1, channels[4], 3)
+        self.final_linear = nn.Sequential(
+            EqualLinear(channels[4] * 4 * 4, channels[4], activation='fused_lrelu'),
+            EqualLinear(channels[4], 1),
+        )
+    def forward(self, input):
+        out = self.convs(input)
+        batch, channel, height, width = out.shape
+        group = min(batch, self.stddev_group)
+        stddev = out.view(
+            group, -1, self.stddev_feat, channel // self.stddev_feat, height, width
+        )
+        stddev = torch.sqrt(stddev.var(0, unbiased=False) + 1e-8)
+        stddev = stddev.mean([2, 3, 4], keepdims=True).squeeze(2)
+        stddev = stddev.repeat(group, 1, height, width)
+        out = torch.cat([out, stddev], 1)
+        out = self.final_conv(out)
+        out = out.view(batch, -1)
+        out = self.final_linear(out)
+        return out

models/stylegan2/op/__init__.py ADDED Viewed

	@@ -0,0 +1,2 @@


1	+ from .fused_act import FusedLeakyReLU, fused_leaky_relu
2	+ from .upfirdn2d import upfirdn2d

models/stylegan2/op/__pycache__/__init__.cpython-39.pyc ADDED Viewed

Binary file (255 Bytes). View file

models/stylegan2/op/__pycache__/fused_act.cpython-39.pyc ADDED Viewed

Binary file (2.83 kB). View file

models/stylegan2/op/fused_act.py ADDED Viewed

	@@ -0,0 +1,86 @@

+import os
+import torch
+from torch import nn
+from torch.autograd import Function
+from torch.utils.cpp_extension import load
+# from . import fused
+module_path = os.path.dirname(__file__)
+fused = load(
+    'fused',
+    sources=[
+        os.path.join(module_path, 'fused_bias_act.cpp'),
+        os.path.join(module_path, 'fused_bias_act_kernel.cu'),
+    ],
+)
+class FusedLeakyReLUFunctionBackward(Function):
+    @staticmethod
+    def forward(ctx, grad_output, out, negative_slope, scale):
+        ctx.save_for_backward(out)
+        ctx.negative_slope = negative_slope
+        ctx.scale = scale
+        empty = grad_output.new_empty(0)
+        grad_input = fused.fused_bias_act(
+            grad_output, empty, out, 3, 1, negative_slope, scale
+        )
+        dim = [0]
+        if grad_input.ndim > 2:
+            dim += list(range(2, grad_input.ndim))
+        grad_bias = grad_input.sum(dim).detach()
+        return grad_input, grad_bias
+    @staticmethod
+    def backward(ctx, gradgrad_input, gradgrad_bias):
+        out, = ctx.saved_tensors
+        gradgrad_out = fused.fused_bias_act(
+            gradgrad_input, gradgrad_bias, out, 3, 1, ctx.negative_slope, ctx.scale
+        )
+        return gradgrad_out, None, None, None
+class FusedLeakyReLUFunction(Function):
+    @staticmethod
+    def forward(ctx, input, bias, negative_slope, scale):
+        empty = input.new_empty(0)
+        out = fused.fused_bias_act(input, bias, empty, 3, 0, negative_slope, scale)
+        ctx.save_for_backward(out)
+        ctx.negative_slope = negative_slope
+        ctx.scale = scale
+        return out
+    @staticmethod
+    def backward(ctx, grad_output):
+        out, = ctx.saved_tensors
+        grad_input, grad_bias = FusedLeakyReLUFunctionBackward.apply(
+            grad_output, out, ctx.negative_slope, ctx.scale
+        )
+        return grad_input, grad_bias, None, None
+class FusedLeakyReLU(nn.Module):
+    def __init__(self, channel, negative_slope=0.2, scale=2 ** 0.5):
+        super().__init__()
+        self.bias = nn.Parameter(torch.zeros(channel))
+        self.negative_slope = negative_slope
+        self.scale = scale
+    def forward(self, input):
+        return fused_leaky_relu(input, self.bias, self.negative_slope, self.scale)
+def fused_leaky_relu(input, bias, negative_slope=0.2, scale=2 ** 0.5):
+    return FusedLeakyReLUFunction.apply(input, bias, negative_slope, scale)

models/stylegan2/op/fused_bias_act.cpp ADDED Viewed

	@@ -0,0 +1,21 @@

+#include <torch/extension.h>
+torch::Tensor fused_bias_act_op(const torch::Tensor& input, const torch::Tensor& bias, const torch::Tensor& refer,
+    int act, int grad, float alpha, float scale);
+#define CHECK_CUDA(x) TORCH_CHECK(x.type().is_cuda(), #x " must be a CUDA tensor")
+#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
+#define CHECK_INPUT(x) CHECK_CUDA(x); CHECK_CONTIGUOUS(x)
+torch::Tensor fused_bias_act(const torch::Tensor& input, const torch::Tensor& bias, const torch::Tensor& refer,
+    int act, int grad, float alpha, float scale) {
+    CHECK_CUDA(input);
+    CHECK_CUDA(bias);
+    return fused_bias_act_op(input, bias, refer, act, grad, alpha, scale);
+}
+PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
+    m.def("fused_bias_act", &fused_bias_act, "fused bias act (CUDA)");
+}

models/stylegan2/op/fused_bias_act_kernel.cu ADDED Viewed

	@@ -0,0 +1,99 @@

+// Copyright (c) 2019, NVIDIA Corporation. All rights reserved.
+//
+// This work is made available under the Nvidia Source Code License-NC.
+// To view a copy of this license, visit
+// https://nvlabs.github.io/stylegan2/license.html
+#include <torch/types.h>
+#include <ATen/ATen.h>
+#include <ATen/AccumulateType.h>
+#include <ATen/cuda/CUDAContext.h>
+#include <ATen/cuda/CUDAApplyUtils.cuh>
+#include <cuda.h>
+#include <cuda_runtime.h>
+template <typename scalar_t>
+static __global__ void fused_bias_act_kernel(scalar_t* out, const scalar_t* p_x, const scalar_t* p_b, const scalar_t* p_ref,
+    int act, int grad, scalar_t alpha, scalar_t scale, int loop_x, int size_x, int step_b, int size_b, int use_bias, int use_ref) {
+    int xi = blockIdx.x * loop_x * blockDim.x + threadIdx.x;
+    scalar_t zero = 0.0;
+    for (int loop_idx = 0; loop_idx < loop_x && xi < size_x; loop_idx++, xi += blockDim.x) {
+        scalar_t x = p_x[xi];
+        if (use_bias) {
+            x += p_b[(xi / step_b) % size_b];
+        }
+        scalar_t ref = use_ref ? p_ref[xi] : zero;
+        scalar_t y;
+        switch (act * 10 + grad) {
+            default:
+            case 10: y = x; break;
+            case 11: y = x; break;
+            case 12: y = 0.0; break;
+            case 30: y = (x > 0.0) ? x : x * alpha; break;
+            case 31: y = (ref > 0.0) ? x : x * alpha; break;
+            case 32: y = 0.0; break;
+        }
+        out[xi] = y * scale;
+    }
+}
+torch::Tensor fused_bias_act_op(const torch::Tensor& input, const torch::Tensor& bias, const torch::Tensor& refer,
+    int act, int grad, float alpha, float scale) {
+    int curDevice = -1;
+    cudaGetDevice(&curDevice);
+    cudaStream_t stream = at::cuda::getCurrentCUDAStream(curDevice);
+    auto x = input.contiguous();
+    auto b = bias.contiguous();
+    auto ref = refer.contiguous();
+    int use_bias = b.numel() ? 1 : 0;
+    int use_ref = ref.numel() ? 1 : 0;
+    int size_x = x.numel();
+    int size_b = b.numel();
+    int step_b = 1;
+    for (int i = 1 + 1; i < x.dim(); i++) {
+        step_b *= x.size(i);
+    }
+    int loop_x = 4;
+    int block_size = 4 * 32;
+    int grid_size = (size_x - 1) / (loop_x * block_size) + 1;
+    auto y = torch::empty_like(x);
+    AT_DISPATCH_FLOATING_TYPES_AND_HALF(x.scalar_type(), "fused_bias_act_kernel", [&] {
+        fused_bias_act_kernel<scalar_t><<<grid_size, block_size, 0, stream>>>(
+            y.data_ptr<scalar_t>(),
+            x.data_ptr<scalar_t>(),
+            b.data_ptr<scalar_t>(),
+            ref.data_ptr<scalar_t>(),
+            act,
+            grad,
+            alpha,
+            scale,
+            loop_x,
+            size_x,
+            step_b,
+            size_b,
+            use_bias,
+            use_ref
+        );
+    });
+    return y;
+}

models/stylegan2/op/upfirdn2d.cpp ADDED Viewed

	@@ -0,0 +1,23 @@

+#include <torch/extension.h>
+torch::Tensor upfirdn2d_op(const torch::Tensor& input, const torch::Tensor& kernel,
+                            int up_x, int up_y, int down_x, int down_y,
+                            int pad_x0, int pad_x1, int pad_y0, int pad_y1);
+#define CHECK_CUDA(x) TORCH_CHECK(x.type().is_cuda(), #x " must be a CUDA tensor")
+#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
+#define CHECK_INPUT(x) CHECK_CUDA(x); CHECK_CONTIGUOUS(x)
+torch::Tensor upfirdn2d(const torch::Tensor& input, const torch::Tensor& kernel,
+                        int up_x, int up_y, int down_x, int down_y,
+                        int pad_x0, int pad_x1, int pad_y0, int pad_y1) {
+    CHECK_CUDA(input);
+    CHECK_CUDA(kernel);
+    return upfirdn2d_op(input, kernel, up_x, up_y, down_x, down_y, pad_x0, pad_x1, pad_y0, pad_y1);
+}
+PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
+    m.def("upfirdn2d", &upfirdn2d, "upfirdn2d (CUDA)");
+}

models/stylegan2/op/upfirdn2d.py ADDED Viewed

	@@ -0,0 +1,186 @@

+import os
+import torch
+from torch.autograd import Function
+from torch.nn import functional as F
+# from . import upfirdn2d as upfirdn2d_op
+from torch.utils.cpp_extension import load
+module_path = os.path.dirname(__file__)
+upfirdn2d_op = load(
+    'upfirdn2d',
+    sources=[
+        os.path.join(module_path, 'upfirdn2d.cpp'),
+        os.path.join(module_path, 'upfirdn2d_kernel.cu'),
+    ],
+)
+class UpFirDn2dBackward(Function):
+    @staticmethod
+    def forward(
+            ctx, grad_output, kernel, grad_kernel, up, down, pad, g_pad, in_size, out_size
+    ):
+        up_x, up_y = up
+        down_x, down_y = down
+        g_pad_x0, g_pad_x1, g_pad_y0, g_pad_y1 = g_pad
+        grad_output = grad_output.reshape(-1, out_size[0], out_size[1], 1)
+        grad_input = upfirdn2d_op.upfirdn2d(
+            grad_output,
+            grad_kernel,
+            down_x,
+            down_y,
+            up_x,
+            up_y,
+            g_pad_x0,
+            g_pad_x1,
+            g_pad_y0,
+            g_pad_y1,
+        )
+        grad_input = grad_input.view(in_size[0], in_size[1], in_size[2], in_size[3])
+        ctx.save_for_backward(kernel)
+        pad_x0, pad_x1, pad_y0, pad_y1 = pad
+        ctx.up_x = up_x
+        ctx.up_y = up_y
+        ctx.down_x = down_x
+        ctx.down_y = down_y
+        ctx.pad_x0 = pad_x0
+        ctx.pad_x1 = pad_x1
+        ctx.pad_y0 = pad_y0
+        ctx.pad_y1 = pad_y1
+        ctx.in_size = in_size
+        ctx.out_size = out_size
+        return grad_input
+    @staticmethod
+    def backward(ctx, gradgrad_input):
+        kernel, = ctx.saved_tensors
+        gradgrad_input = gradgrad_input.reshape(-1, ctx.in_size[2], ctx.in_size[3], 1)
+        gradgrad_out = upfirdn2d_op.upfirdn2d(
+            gradgrad_input,
+            kernel,
+            ctx.up_x,
+            ctx.up_y,
+            ctx.down_x,
+            ctx.down_y,
+            ctx.pad_x0,
+            ctx.pad_x1,
+            ctx.pad_y0,
+            ctx.pad_y1,
+        )
+        # gradgrad_out = gradgrad_out.view(ctx.in_size[0], ctx.out_size[0], ctx.out_size[1], ctx.in_size[3])
+        gradgrad_out = gradgrad_out.view(
+            ctx.in_size[0], ctx.in_size[1], ctx.out_size[0], ctx.out_size[1]
+        )
+        return gradgrad_out, None, None, None, None, None, None, None, None
+class UpFirDn2d(Function):
+    @staticmethod
+    def forward(ctx, input, kernel, up, down, pad):
+        up_x, up_y = up
+        down_x, down_y = down
+        pad_x0, pad_x1, pad_y0, pad_y1 = pad
+        kernel_h, kernel_w = kernel.shape
+        batch, channel, in_h, in_w = input.shape
+        ctx.in_size = input.shape
+        input = input.reshape(-1, in_h, in_w, 1)
+        ctx.save_for_backward(kernel, torch.flip(kernel, [0, 1]))
+        out_h = (in_h * up_y + pad_y0 + pad_y1 - kernel_h) // down_y + 1
+        out_w = (in_w * up_x + pad_x0 + pad_x1 - kernel_w) // down_x + 1
+        ctx.out_size = (out_h, out_w)
+        ctx.up = (up_x, up_y)
+        ctx.down = (down_x, down_y)
+        ctx.pad = (pad_x0, pad_x1, pad_y0, pad_y1)
+        g_pad_x0 = kernel_w - pad_x0 - 1
+        g_pad_y0 = kernel_h - pad_y0 - 1
+        g_pad_x1 = in_w * up_x - out_w * down_x + pad_x0 - up_x + 1
+        g_pad_y1 = in_h * up_y - out_h * down_y + pad_y0 - up_y + 1
+        ctx.g_pad = (g_pad_x0, g_pad_x1, g_pad_y0, g_pad_y1)
+        out = upfirdn2d_op.upfirdn2d(
+            input, kernel, up_x, up_y, down_x, down_y, pad_x0, pad_x1, pad_y0, pad_y1
+        )
+        # out = out.view(major, out_h, out_w, minor)
+        out = out.view(-1, channel, out_h, out_w)
+        return out
+    @staticmethod
+    def backward(ctx, grad_output):
+        kernel, grad_kernel = ctx.saved_tensors
+        grad_input = UpFirDn2dBackward.apply(
+            grad_output,
+            kernel,
+            grad_kernel,
+            ctx.up,
+            ctx.down,
+            ctx.pad,
+            ctx.g_pad,
+            ctx.in_size,
+            ctx.out_size,
+        )
+        return grad_input, None, None, None, None
+def upfirdn2d(input, kernel, up=1, down=1, pad=(0, 0)):
+    out = UpFirDn2d.apply(
+        input, kernel, (up, up), (down, down), (pad[0], pad[1], pad[0], pad[1])
+    )
+    return out
+def upfirdn2d_native(
+        input, kernel, up_x, up_y, down_x, down_y, pad_x0, pad_x1, pad_y0, pad_y1
+):
+    _, in_h, in_w, minor = input.shape
+    kernel_h, kernel_w = kernel.shape
+    out = input.view(-1, in_h, 1, in_w, 1, minor)
+    out = F.pad(out, [0, 0, 0, up_x - 1, 0, 0, 0, up_y - 1])
+    out = out.view(-1, in_h * up_y, in_w * up_x, minor)
+    out = F.pad(
+        out, [0, 0, max(pad_x0, 0), max(pad_x1, 0), max(pad_y0, 0), max(pad_y1, 0)]
+    )
+    out = out[
+          :,
+          max(-pad_y0, 0): out.shape[1] - max(-pad_y1, 0),
+          max(-pad_x0, 0): out.shape[2] - max(-pad_x1, 0),
+          :,
+          ]
+    out = out.permute(0, 3, 1, 2)
+    out = out.reshape(
+        [-1, 1, in_h * up_y + pad_y0 + pad_y1, in_w * up_x + pad_x0 + pad_x1]
+    )
+    w = torch.flip(kernel, [0, 1]).view(1, 1, kernel_h, kernel_w)
+    out = F.conv2d(out, w)
+    out = out.reshape(
+        -1,
+        minor,
+        in_h * up_y + pad_y0 + pad_y1 - kernel_h + 1,
+        in_w * up_x + pad_x0 + pad_x1 - kernel_w + 1,
+    )
+    out = out.permute(0, 2, 3, 1)
+    return out[:, ::down_y, ::down_x, :]

models/stylegan2/op/upfirdn2d_kernel.cu ADDED Viewed

	@@ -0,0 +1,272 @@

+// Copyright (c) 2019, NVIDIA Corporation. All rights reserved.
+//
+// This work is made available under the Nvidia Source Code License-NC.
+// To view a copy of this license, visit
+// https://nvlabs.github.io/stylegan2/license.html
+#include <torch/types.h>
+#include <ATen/ATen.h>
+#include <ATen/AccumulateType.h>
+#include <ATen/cuda/CUDAContext.h>
+#include <ATen/cuda/CUDAApplyUtils.cuh>
+#include <cuda.h>
+#include <cuda_runtime.h>
+static __host__ __device__ __forceinline__ int floor_div(int a, int b) {
+    int c = a / b;
+    if (c * b > a) {
+        c--;
+    }
+    return c;
+}
+struct UpFirDn2DKernelParams {
+    int up_x;
+    int up_y;
+    int down_x;
+    int down_y;
+    int pad_x0;
+    int pad_x1;
+    int pad_y0;
+    int pad_y1;
+    int major_dim;
+    int in_h;
+    int in_w;
+    int minor_dim;
+    int kernel_h;
+    int kernel_w;
+    int out_h;
+    int out_w;
+    int loop_major;
+    int loop_x;
+};
+template <typename scalar_t, int up_x, int up_y, int down_x, int down_y, int kernel_h, int kernel_w, int tile_out_h, int tile_out_w>
+__global__ void upfirdn2d_kernel(scalar_t* out, const scalar_t* input, const scalar_t* kernel, const UpFirDn2DKernelParams p) {
+    const int tile_in_h = ((tile_out_h - 1) * down_y + kernel_h - 1) / up_y + 1;
+    const int tile_in_w = ((tile_out_w - 1) * down_x + kernel_w - 1) / up_x + 1;
+    __shared__ volatile float sk[kernel_h][kernel_w];
+    __shared__ volatile float sx[tile_in_h][tile_in_w];
+    int minor_idx = blockIdx.x;
+    int tile_out_y = minor_idx / p.minor_dim;
+    minor_idx -= tile_out_y * p.minor_dim;
+    tile_out_y *= tile_out_h;
+    int tile_out_x_base = blockIdx.y * p.loop_x * tile_out_w;
+    int major_idx_base = blockIdx.z * p.loop_major;
+    if (tile_out_x_base >= p.out_w | tile_out_y >= p.out_h | major_idx_base >= p.major_dim) {
+        return;
+    }
+    for (int tap_idx = threadIdx.x; tap_idx < kernel_h * kernel_w; tap_idx += blockDim.x) {
+        int ky = tap_idx / kernel_w;
+        int kx = tap_idx - ky * kernel_w;
+        scalar_t v = 0.0;
+        if (kx < p.kernel_w & ky < p.kernel_h) {
+            v = kernel[(p.kernel_h - 1 - ky) * p.kernel_w + (p.kernel_w - 1 - kx)];
+        }
+        sk[ky][kx] = v;
+    }
+    for (int loop_major = 0, major_idx = major_idx_base; loop_major < p.loop_major & major_idx < p.major_dim; loop_major++, major_idx++) {
+        for (int loop_x = 0, tile_out_x = tile_out_x_base; loop_x < p.loop_x & tile_out_x < p.out_w; loop_x++, tile_out_x += tile_out_w) {
+            int tile_mid_x = tile_out_x * down_x + up_x - 1 - p.pad_x0;
+            int tile_mid_y = tile_out_y * down_y + up_y - 1 - p.pad_y0;
+            int tile_in_x = floor_div(tile_mid_x, up_x);
+            int tile_in_y = floor_div(tile_mid_y, up_y);
+            __syncthreads();
+            for (int in_idx = threadIdx.x; in_idx < tile_in_h * tile_in_w; in_idx += blockDim.x) {
+                int rel_in_y = in_idx / tile_in_w;
+                int rel_in_x = in_idx - rel_in_y * tile_in_w;
+                int in_x = rel_in_x + tile_in_x;
+                int in_y = rel_in_y + tile_in_y;
+                scalar_t v = 0.0;
+                if (in_x >= 0 & in_y >= 0 & in_x < p.in_w & in_y < p.in_h) {
+                    v = input[((major_idx * p.in_h + in_y) * p.in_w + in_x) * p.minor_dim + minor_idx];
+                }
+                sx[rel_in_y][rel_in_x] = v;
+            }
+            __syncthreads();
+            for (int out_idx = threadIdx.x; out_idx < tile_out_h * tile_out_w; out_idx += blockDim.x) {
+                int rel_out_y = out_idx / tile_out_w;
+                int rel_out_x = out_idx - rel_out_y * tile_out_w;
+                int out_x = rel_out_x + tile_out_x;
+                int out_y = rel_out_y + tile_out_y;
+                int mid_x = tile_mid_x + rel_out_x * down_x;
+                int mid_y = tile_mid_y + rel_out_y * down_y;
+                int in_x = floor_div(mid_x, up_x);
+                int in_y = floor_div(mid_y, up_y);
+                int rel_in_x = in_x - tile_in_x;
+                int rel_in_y = in_y - tile_in_y;
+                int kernel_x = (in_x + 1) * up_x - mid_x - 1;
+                int kernel_y = (in_y + 1) * up_y - mid_y - 1;
+                scalar_t v = 0.0;
+                #pragma unroll
+                for (int y = 0; y < kernel_h / up_y; y++)
+                    #pragma unroll
+                    for (int x = 0; x < kernel_w / up_x; x++)
+                        v += sx[rel_in_y + y][rel_in_x + x] * sk[kernel_y + y * up_y][kernel_x + x * up_x];
+                if (out_x < p.out_w & out_y < p.out_h) {
+                    out[((major_idx * p.out_h + out_y) * p.out_w + out_x) * p.minor_dim + minor_idx] = v;
+                }
+            }
+        }
+    }
+}
+torch::Tensor upfirdn2d_op(const torch::Tensor& input, const torch::Tensor& kernel,
+    int up_x, int up_y, int down_x, int down_y,
+    int pad_x0, int pad_x1, int pad_y0, int pad_y1) {
+    int curDevice = -1;
+    cudaGetDevice(&curDevice);
+    cudaStream_t stream = at::cuda::getCurrentCUDAStream(curDevice);
+    UpFirDn2DKernelParams p;
+    auto x = input.contiguous();
+    auto k = kernel.contiguous();
+    p.major_dim = x.size(0);
+    p.in_h = x.size(1);
+    p.in_w = x.size(2);
+    p.minor_dim = x.size(3);
+    p.kernel_h = k.size(0);
+    p.kernel_w = k.size(1);
+    p.up_x = up_x;
+    p.up_y = up_y;
+    p.down_x = down_x;
+    p.down_y = down_y;
+    p.pad_x0 = pad_x0;
+    p.pad_x1 = pad_x1;
+    p.pad_y0 = pad_y0;
+    p.pad_y1 = pad_y1;
+    p.out_h = (p.in_h * p.up_y + p.pad_y0 + p.pad_y1 - p.kernel_h + p.down_y) / p.down_y;
+    p.out_w = (p.in_w * p.up_x + p.pad_x0 + p.pad_x1 - p.kernel_w + p.down_x) / p.down_x;
+    auto out = at::empty({p.major_dim, p.out_h, p.out_w, p.minor_dim}, x.options());
+    int mode = -1;
+    int tile_out_h;
+    int tile_out_w;
+    if (p.up_x == 1 && p.up_y == 1 && p.down_x == 1 && p.down_y == 1 && p.kernel_h <= 4 && p.kernel_w <= 4) {
+        mode = 1;
+        tile_out_h = 16;
+        tile_out_w = 64;
+    }
+    if (p.up_x == 1 && p.up_y == 1 && p.down_x == 1 && p.down_y == 1 && p.kernel_h <= 3 && p.kernel_w <= 3) {
+        mode = 2;
+        tile_out_h = 16;
+        tile_out_w = 64;
+    }
+    if (p.up_x == 2 && p.up_y == 2 && p.down_x == 1 && p.down_y == 1 && p.kernel_h <= 4 && p.kernel_w <= 4) {
+        mode = 3;
+        tile_out_h = 16;
+        tile_out_w = 64;
+    }
+    if (p.up_x == 2 && p.up_y == 2 && p.down_x == 1 && p.down_y == 1 && p.kernel_h <= 2 && p.kernel_w <= 2) {
+        mode = 4;
+        tile_out_h = 16;
+        tile_out_w = 64;
+    }
+    if (p.up_x == 1 && p.up_y == 1 && p.down_x == 2 && p.down_y == 2 && p.kernel_h <= 4 && p.kernel_w <= 4) {
+        mode = 5;
+        tile_out_h = 8;
+        tile_out_w = 32;
+    }
+    if (p.up_x == 1 && p.up_y == 1 && p.down_x == 2 && p.down_y == 2 && p.kernel_h <= 2 && p.kernel_w <= 2) {
+        mode = 6;
+        tile_out_h = 8;
+        tile_out_w = 32;
+    }
+    dim3 block_size;
+    dim3 grid_size;
+    if (tile_out_h > 0 && tile_out_w) {
+        p.loop_major = (p.major_dim - 1) / 16384 + 1;
+        p.loop_x = 1;
+        block_size = dim3(32 * 8, 1, 1);
+        grid_size = dim3(((p.out_h - 1) / tile_out_h + 1) * p.minor_dim,
+                         (p.out_w - 1) / (p.loop_x * tile_out_w) + 1,
+                         (p.major_dim - 1) / p.loop_major + 1);
+    }
+    AT_DISPATCH_FLOATING_TYPES_AND_HALF(x.scalar_type(), "upfirdn2d_cuda", [&] {
+        switch (mode) {
+        case 1:
+            upfirdn2d_kernel<scalar_t, 1, 1, 1, 1, 4, 4, 16, 64><<<grid_size, block_size, 0, stream>>>(
+                out.data_ptr<scalar_t>(), x.data_ptr<scalar_t>(), k.data_ptr<scalar_t>(), p
+            );
+            break;
+        case 2:
+            upfirdn2d_kernel<scalar_t, 1, 1, 1, 1, 3, 3, 16, 64><<<grid_size, block_size, 0, stream>>>(
+                out.data_ptr<scalar_t>(), x.data_ptr<scalar_t>(), k.data_ptr<scalar_t>(), p
+            );
+            break;
+        case 3:
+            upfirdn2d_kernel<scalar_t, 2, 2, 1, 1, 4, 4, 16, 64><<<grid_size, block_size, 0, stream>>>(
+                out.data_ptr<scalar_t>(), x.data_ptr<scalar_t>(), k.data_ptr<scalar_t>(), p
+            );
+            break;
+        case 4:
+            upfirdn2d_kernel<scalar_t, 2, 2, 1, 1, 2, 2, 16, 64><<<grid_size, block_size, 0, stream>>>(
+                out.data_ptr<scalar_t>(), x.data_ptr<scalar_t>(), k.data_ptr<scalar_t>(), p
+            );
+            break;
+        case 5:
+            upfirdn2d_kernel<scalar_t, 1, 1, 2, 2, 4, 4, 8, 32><<<grid_size, block_size, 0, stream>>>(
+                out.data_ptr<scalar_t>(), x.data_ptr<scalar_t>(), k.data_ptr<scalar_t>(), p
+            );
+            break;
+        case 6:
+            upfirdn2d_kernel<scalar_t, 1, 1, 2, 2, 4, 4, 8, 32><<<grid_size, block_size, 0, stream>>>(
+                out.data_ptr<scalar_t>(), x.data_ptr<scalar_t>(), k.data_ptr<scalar_t>(), p
+            );
+            break;
+        }
+    });
+    return out;
+}

models/stylegene/__init__.py ADDED Viewed

File without changes

models/stylegene/__pycache__/__init__.cpython-39.pyc ADDED Viewed

Binary file (145 Bytes). View file

models/stylegene/__pycache__/api.cpython-39.pyc ADDED Viewed

Binary file (3.28 kB). View file

models/stylegene/api.py ADDED Viewed

	@@ -0,0 +1,94 @@

+import torch
+import numpy as np
+import torch.nn.functional as F
+from models.stylegan2.model import Generator
+from models.encoders.psp_encoders import Encoder4Editing
+from models.stylegene.model import MappingSub2W, MappingW2Sub
+from models.stylegene.util import get_keys, requires_grad, load_img
+from models.stylegene.gene_pool import GenePoolFactory
+from models.stylegene.gene_crossover_mutation import fuse_latent
+from models.stylegene.fair_face_model import init_fair_model, predict_race
+from configs import path_ckpt_e4e, path_ckpt_stylegan2, path_ckpt_stylegene, path_ckpt_genepool, path_dataset_ffhq
+from preprocess.align_images import align_face
+device = torch.device('cuda:0') if torch.cuda.is_available() else torch.device('cpu')
+def init_model(image_size=1024, latent_dim=512):
+    ckp = torch.load(path_ckpt_e4e, map_location='cpu')
+    encoder = Encoder4Editing(50, 'ir_se', image_size).eval()
+    encoder.load_state_dict(get_keys(ckp, 'encoder'), strict=True)
+    mean_latent = ckp['latent_avg'].to('cpu')
+    mean_latent.unsqueeze_(0)
+    generator = Generator(image_size, latent_dim, 8)
+    checkpoint = torch.load(path_ckpt_stylegan2, map_location='cpu')
+    generator.load_state_dict(checkpoint["g_ema"], strict=False)
+    generator.eval()
+    sub2w = MappingSub2W(N=18).eval()
+    w2sub34 = MappingW2Sub(N=18).eval()
+    ckp = torch.load(path_ckpt_stylegene, map_location='cpu')
+    w2sub34.load_state_dict(get_keys(ckp, 'w2sub34'))
+    sub2w.load_state_dict(get_keys(ckp, 'sub2w'))
+    requires_grad(sub2w, False)
+    requires_grad(w2sub34, False)
+    requires_grad(encoder, False)
+    requires_grad(generator, False)
+    return encoder, generator, sub2w, w2sub34, mean_latent
+# init model
+encoder, generator, sub2w, w2sub34, mean_latent = init_model()
+encoder, generator, sub2w, w2sub34, mean_latent = encoder.to(device), generator.to(device), sub2w.to(
+    device), w2sub34.to(device), mean_latent.to(device)
+model_fair_7 = init_fair_model(device)  # init FairFace model
+# load a GenePool
+geneFactor = GenePoolFactory(root_ffhq=path_dataset_ffhq, device=device, mean_latent=mean_latent, max_sample=300)
+geneFactor.pools = torch.load(path_ckpt_genepool)
+print("gene pool loaded!")
+def tensor2rgb(tensor):
+    tensor = (tensor * 0.5 + 0.5) * 255
+    tensor = torch.clip(tensor, 0, 255).squeeze(0)
+    tensor = tensor.detach().cpu().numpy().transpose(1, 2, 0)
+    tensor = tensor.astype(np.uint8)
+    return tensor
+def generate_child(w18_F, w18_M, random_fakes, gamma=0.46, eta=0.4):
+    w18_syn = fuse_latent(w2sub34, sub2w, w18_F=w18_F, w18_M=w18_M,
+                          random_fakes=random_fakes, fixed_gamma=gamma, fixed_eta=eta)
+    img_C, _ = generator([w18_syn], return_latents=True, input_is_latent=True)
+    return img_C, w18_syn
+def synthesize_descendant(pF, pM, attributes=None):
+    gender_all = ['male', 'female']
+    ages_all = ['0-2', '3-9', '10-19', '20-29', '30-39', '40-49', '50-59', '60-69', '70+']
+    if attributes is None:
+        attributes = {'age': ages_all[0], 'gender': gender_all[0], 'gamma': 0.47, 'eta': 0.4}
+    imgF = align_face(pF)
+    imgM = align_face(pM)
+    imgF = load_img(imgF)
+    imgM = load_img(imgM)
+    imgF, imgM = imgF.to(device), imgM.to(device)
+    father_race, _, _, _ = predict_race(model_fair_7, imgF.clone(), imgF.device)
+    mother_race, _, _, _ = predict_race(model_fair_7, imgM.clone(), imgM.device)
+    w18_1 = encoder(F.interpolate(imgF, size=(256, 256))) + mean_latent
+    w18_2 = encoder(F.interpolate(imgM, size=(256, 256))) + mean_latent
+    random_fakes = []
+    for r in list({father_race, mother_race}):  # search RFGs from Gene Pool
+        random_fakes = random_fakes + geneFactor(encoder, w2sub34, attributes['age'], attributes['gender'], r)
+    img_C, w18_syn = generate_child(w18_1.clone(), w18_2.clone(), random_fakes,
+                                    gamma=attributes['gamma'], eta=attributes['eta'])
+    img_C = tensor2rgb(img_C)
+    img_F = tensor2rgb(imgF)
+    img_M = tensor2rgb(imgM)
+    return img_F, img_M, img_C

models/stylegene/data_util.py ADDED Viewed

	@@ -0,0 +1,36 @@

+"""
+Copyright (C) 2021 NVIDIA Corporation.  All rights reserved.
+Licensed under The MIT License (MIT)
+Permission is hereby granted, free of charge, to any person obtaining a copy of
+this software and associated documentation files (the "Software"), to deal in
+the Software without restriction, including without limitation the rights to
+use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
+the Software, and to permit persons to whom the Software is furnished to do so,
+subject to the following conditions:
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
+FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
+COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
+IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
+CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+"""
+face_class = ['background', 'head', 'head***cheek', 'head***chin', 'head***ear', 'head***ear***helix',
+              'head***ear***lobule', 'head***eye***botton lid', 'head***eye***eyelashes', 'head***eye***iris',
+              'head***eye***pupil', 'head***eye***sclera', 'head***eye***tear duct', 'head***eye***top lid',
+              'head***eyebrow', 'head***forehead', 'head***frown', 'head***hair', 'head***hair***sideburns',
+              'head***jaw', 'head***moustache', 'head***mouth***inferior lip', 'head***mouth***oral comisure',
+              'head***mouth***superior lip', 'head***mouth***teeth', 'head***neck', 'head***nose',
+              'head***nose***ala of nose', 'head***nose***bridge', 'head***nose***nose tip', 'head***nose***nostril',
+              'head***philtrum', 'head***temple', 'head***wrinkles']
+face_must = ['head***forehead', 'head***frown', 'head***nose***bridge', 'head***nose', 'head***nose***ala of nose',
+             'head***nose***nose tip', 'head***nose***nostril', 'head***mouth***inferior lip',
+             'head***mouth***superior lip', 'head***chin', 'head***eye***top lid', 'head***eye***pupil',
+             'head***eye***iris', 'head***eye***tear duct']
+face_shape = ['head', 'head***chin', 'head***jaw', 'head***moustache', 'head***cheek']

models/stylegene/fair_face_model.py ADDED Viewed

	@@ -0,0 +1,61 @@

+import torch
+import numpy as np
+from PIL import Image
+from torch import nn
+import torchvision
+import torch.nn.functional as F
+from torchvision import transforms
+from configs import path_ckpt_fairface
+# code adapted from https://github.com/dchen236/FairFace
+def init_fair_model(device, path_ckpt=None):
+    if path_ckpt is None:
+        path_ckpt = path_ckpt_fairface
+    model_fair_7 = torchvision.models.resnet34(pretrained=False)
+    model_fair_7.fc = nn.Linear(model_fair_7.fc.in_features, 18)
+    model_fair_7.load_state_dict(
+        torch.load(path_ckpt))
+    model_fair_7 = model_fair_7.to(device)
+    model_fair_7.eval()
+    return model_fair_7
+def predict_race(model_fair_7, path_img, device):
+    if type(path_img) == str:
+        trans = transforms.Compose([
+            transforms.Resize((224, 224)),
+            transforms.ToTensor(),
+            transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])
+        ])
+        image = Image.open(path_img)
+        image = trans(image)
+        image = image.view(1, 3, 224, 224)  # reshape image to match model dimensions (1 batch size)
+    elif type(path_img) == torch.Tensor:
+        trans = transforms.Compose([
+            transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])
+        ])
+        image = F.interpolate(path_img, (224, 224))
+        image = image * 0.5 + 0.5
+        image = trans(image)
+        image = image.view(1, 3, 224, 224)
+    image = image.to(device)
+    outputs = model_fair_7(image)
+    outputs = outputs.cpu().detach().numpy()
+    outputs = np.squeeze(outputs)
+    race_outputs = outputs[:7]
+    gender_outputs = outputs[7:9]
+    age_outputs = outputs[9:18]
+    race_score = np.exp(race_outputs) / np.sum(np.exp(race_outputs))
+    gender_score = np.exp(gender_outputs) / np.sum(np.exp(gender_outputs))
+    age_score = np.exp(age_outputs) / np.sum(np.exp(age_outputs))
+    race_pred = np.argmax(race_score)
+    gender_pred = np.argmax(gender_score)
+    age_pred = np.argmax(age_score)
+    race_label = ['White', 'Black', 'Latino_Hispanic', 'East Asian', 'Southeast Asian', 'Indian', 'Middle Eastern']
+    return race_label[race_pred], race_pred, gender_pred, age_pred

models/stylegene/gene_crossover_mutation.py ADDED Viewed

	@@ -0,0 +1,64 @@

+import torch
+from .data_util import face_class, face_shape
+import random
+def reparameterize(mu, logvar):
+    """
+    Reparameterization trick to sample from N(mu, var) from
+    N(0,1).
+    :param mu: (Tensor) Mean of the latent Gaussian [B x D]
+    :param logvar: (Tensor) Standard deviation of the latent Gaussian [B x D]
+    :return: (Tensor) [B x D]
+    """
+    std = torch.exp(0.5 * logvar)
+    eps = torch.randn_like(std)
+    return eps * std + mu
+def mix(w18_F, w18_M, w18_syn):
+    for k in [8, 9, 10, 11, 12, 13, 14, 15, 16, 17]:
+        w18_syn[:, k, :] = w18_F[:, k, :] * 0.5 + w18_M[:, k, :] * 0.5
+    return w18_syn
+def fuse_latent(w2sub34, sub2w, w18_F, w18_M, random_fakes, fixed_gamma=0.47, fixed_eta=0.4):
+    device = w18_F.device
+    mu_F, var_F, sub34_F = w2sub34(w18_F)
+    mu_M, var_M, sub34_M = w2sub34(w18_M)
+    new_sub34 = torch.zeros_like(sub34_F, dtype=torch.float, device=device)
+    if len(random_fakes) == 0:  # EXCEPTION HANDLER (No matching gene pool)
+        random_fakes = [(mu_F.cpu(), var_F.cpu())] + [(mu_M.cpu(), var_M.cpu())]
+    # region genetic variation weights
+    weights = {}
+    for i in face_class:
+        weights[i] = (random.uniform(0, 1 - float(fixed_gamma)), float(fixed_gamma))
+    # select genetic regions
+    cur_class = random.sample(face_class, int(len(face_class) * (1 - float(fixed_eta))))
+    for i, classname in enumerate(face_class):
+        if classname == 'background':
+            new_sub34[:, :, i, :] = reparameterize(mu_F[:, :, i, :], var_F[:, :, i, :])
+            continue
+        if classname in cur_class:  # # corresponding to t = 0 in Eq.10
+            fake_mu, fake_var = random.choice(random_fakes)
+            w_i, b_i = weights[classname]
+            new_sub34[:, :, i, :] = reparameterize(
+                mu_F[:, :, i, :] * w_i + fake_mu[:, :, i, :].to(device) * b_i + mu_M[:, :, i, :] * (1 - w_i - b_i),
+                var_F[:, :, i, :] * w_i + fake_var[:, :, i, :].to(device) * b_i + var_M[:, :, i, :] * (1 - w_i - b_i))
+        else:  # corresponding to t = 1 in Eq.10
+            fake_mu, fake_var = random.choice(random_fakes)
+            fake_latent = reparameterize(fake_mu[:, :, i, :], fake_var[:, :, i, :]).to(device)
+            var = fake_latent
+            new_sub34[:, :, i, :] = new_sub34[:, :, i, :] + var
+    w18_syn = sub2w(new_sub34)
+    w18_syn = mix(w18_F, w18_M, w18_syn)
+    return w18_syn

models/stylegene/gene_pool.py ADDED Viewed

	@@ -0,0 +1,42 @@

+import os
+import random
+import pandas as pd
+import torch.nn.functional as F
+from .util import load_img
+from configs import path_csv_ffhq_attritube
+class GenePoolFactory(object):
+    def __init__(self, root_ffhq, device, mean_latent, max_sample=100):
+        self.device = device
+        self.mean_latent = mean_latent
+        self.root_ffhq = root_ffhq
+        self.max_sample = max_sample
+        self.pools = {}
+        path_ffhq_attributes = path_csv_ffhq_attritube
+        self.df = pd.read_csv(path_ffhq_attributes)
+        self.df.replace('Male', 'male', inplace=True)
+        self.df.replace('Female', 'female', inplace=True)
+    def __call__(self, encoder, w2sub34, age, gender, race):
+        keyname = f'{age}-{gender}-{race}'
+        if keyname in self.pools.keys():
+            return self.pools[keyname]
+        elif self.root_ffhq is not None:
+            result = self.df.query(f'gender == "{gender}" and age == "{age}" and race == "{race}"')
+            result = result[['file_id']].values
+            tmp = []
+            random.shuffle(result)
+            for fid in result[:self.max_sample]:
+                filename = format(int(fid[0]), '05d') + ".png"
+                img = load_img(os.path.join(self.root_ffhq, filename))
+                img = img.to(self.device)
+                w18_1 = encoder(F.interpolate(img, size=(256, 256))) + self.mean_latent
+                mu, var, sub34_1 = w2sub34(w18_1)
+                tmp.append((mu.cpu(), var.cpu()))
+            self.pools[keyname] = tmp
+            return self.pools[keyname]
+        else:
+            return []

models/stylegene/model.py ADDED Viewed

	@@ -0,0 +1,209 @@

+import torch
+from torch import nn
+from functools import partial
+from einops.layers.torch import Rearrange, Reduce
+from einops import rearrange
+pair = lambda x: x if isinstance(x, tuple) else (x, x)
+class PreNormResidual(nn.Module):
+    def __init__(self, dim, fn):
+        super().__init__()
+        self.fn = fn
+        self.norm = nn.LayerNorm(dim)
+    def forward(self, x):
+        return self.fn(self.norm(x)) + x
+def FeedForward(dim, expansion_factor=4, dropout=0., dense=nn.Linear):
+    inner_dim = int(dim * expansion_factor)
+    return nn.Sequential(
+        dense(dim, inner_dim),
+        nn.GELU(),
+        nn.Dropout(dropout),
+        dense(inner_dim, dim),
+        nn.Dropout(dropout)
+    )
+class MappingSub2W(nn.Module):
+    def __init__(self, N=8, dim=512, depth=6, expansion_factor=4., expansion_factor_token=0.5, dropout=0.1):
+        super(MappingSub2W, self).__init__()
+        num_patches = N * 34
+        chan_first, chan_last = partial(nn.Conv1d, kernel_size=1), nn.Linear
+        self.layer = nn.Sequential(
+            Rearrange('b c h w -> b (c h) w'),
+            *[nn.Sequential(
+                PreNormResidual(dim, FeedForward(num_patches, expansion_factor, dropout, chan_first)),
+                PreNormResidual(dim, FeedForward(dim, expansion_factor_token, dropout, chan_last))
+            ) for _ in range(depth)],
+            nn.LayerNorm(dim),
+            Rearrange('b c h -> b h c'),
+            nn.Linear(34 * N, 34 * N),
+            nn.LayerNorm(34 * N),
+            nn.GELU(),
+            nn.Linear(34 * N, N),
+            Rearrange('b h c -> b c h')
+        )
+    def forward(self, x):
+        return self.layer(x)
+class MappingW2Sub(nn.Module):
+    def __init__(self, N=8, dim=512, depth=8, expansion_factor=4., expansion_factor_token=0.5, dropout=0.1):
+        super(MappingW2Sub, self).__init__()
+        self.N = N
+        num_patches = N * 34
+        chan_first, chan_last = partial(nn.Conv1d, kernel_size=1), nn.Linear
+        self.layer = nn.Sequential(
+            Rearrange('b c h -> b h c'),
+            nn.Linear(N, num_patches),
+            Rearrange('b h c -> b c h'),
+            *[nn.Sequential(
+                PreNormResidual(dim, FeedForward(num_patches, expansion_factor, dropout, chan_first)),
+                PreNormResidual(dim, FeedForward(dim, expansion_factor_token, dropout, chan_last))
+            ) for _ in range(depth)],
+            nn.LayerNorm(dim)
+        )
+        self.mu_fc = nn.Sequential(
+            *[nn.Sequential(
+                PreNormResidual(dim, FeedForward(num_patches, expansion_factor, dropout, chan_first)),
+                PreNormResidual(dim, FeedForward(dim, expansion_factor_token, dropout, chan_last))
+            ) for _ in range(2)],
+            nn.LayerNorm(dim),
+            nn.Tanh(),
+            Rearrange('b c h -> b (c h)')
+        )
+        self.var_fc = nn.Sequential(
+            *[nn.Sequential(
+                PreNormResidual(dim, FeedForward(num_patches, expansion_factor, dropout, chan_first)),
+                PreNormResidual(dim, FeedForward(dim, expansion_factor_token, dropout, chan_last))
+            ) for _ in range(2)],
+            nn.LayerNorm(dim),
+            nn.Tanh(),
+            Rearrange('b c h -> b (c h)')
+        )
+    def reparameterize(self, mu, logvar):
+        """
+        Reparameterization trick to sample from N(mu, var) from
+        N(0,1).
+        :param mu: (Tensor) Mean of the latent Gaussian [B x D]
+        :param logvar: (Tensor) Standard deviation of the latent Gaussian [B x D]
+        :return: (Tensor) [B x D]
+        """
+        std = torch.exp(0.5 * logvar)
+        eps = torch.randn_like(std)
+        return eps * std + mu
+    def forward(self, x):
+        f = self.layer(x)
+        mu = self.mu_fc(f)
+        var = self.var_fc(f)
+        z = self.reparameterize(mu, var)
+        z = rearrange(z, 'a (b c d) -> a b c d', b=self.N, c=34)
+        return rearrange(mu, 'a (b c d) -> a b c d', b=self.N, c=34), rearrange(var, 'a (b c d) -> a b c d',
+                                                                                b=self.N, c=34), z
+class HeadEncoder(nn.Module):
+    def __init__(self, N=8, dim=512, depth=2, expansion_factor=4., expansion_factor_token=0.5, dropout=0.1):
+        super(HeadEncoder, self).__init__()
+        channels = [32, 64, 64, 64]
+        self.N = N
+        num_patches = N
+        chan_first, chan_last = partial(nn.Conv1d, kernel_size=1), nn.Linear
+        self.s1 = nn.Sequential(
+            nn.Conv2d(channels[0], channels[1], kernel_size=5, padding=2, stride=2),
+            nn.BatchNorm2d(channels[1]),
+            nn.LeakyReLU(),
+            nn.Conv2d(channels[1], channels[2], kernel_size=5, padding=2, stride=2),
+            nn.BatchNorm2d(channels[2]),
+            nn.LeakyReLU(),
+            nn.Conv2d(channels[2], channels[3], kernel_size=5, padding=2, stride=2),
+            nn.BatchNorm2d(channels[3]),
+            nn.LeakyReLU())
+        self.mlp1 = nn.Linear(channels[3] * 8 * 8, 512)
+        self.up_N = nn.Linear(1, N)
+        self.mu_fc = nn.Sequential(
+            *[nn.Sequential(
+                PreNormResidual(dim, FeedForward(num_patches, expansion_factor, dropout, chan_first)),
+                PreNormResidual(dim, FeedForward(dim, expansion_factor_token, dropout, chan_last))
+            ) for _ in range(depth)],
+            nn.LayerNorm(dim),
+            nn.Tanh()
+        )
+        self.var_fc = nn.Sequential(
+            *[nn.Sequential(
+                PreNormResidual(dim, FeedForward(num_patches, expansion_factor, dropout, chan_first)),
+                PreNormResidual(dim, FeedForward(dim, expansion_factor_token, dropout, chan_last))
+            ) for _ in range(depth)],
+            nn.LayerNorm(dim),
+            nn.Tanh()
+        )
+    def reparameterize(self, mu, logvar):
+        """
+        Reparameterization trick to sample from N(mu, var) from
+        N(0,1).
+        :param mu: (Tensor) Mean of the latent Gaussian [B x D]
+        :param logvar: (Tensor) Standard deviation of the latent Gaussian [B x D]
+        :return: (Tensor) [B x D]
+        """
+        std = torch.exp(0.5 * logvar)
+        eps = torch.randn_like(std)
+        return eps * std + mu
+    def forward(self, x):
+        feature = self.s1(x)
+        s2 = torch.flatten(feature, start_dim=1)
+        s2 = self.mlp1(s2).unsqueeze(2)
+        s2 = self.up_N(s2)
+        s2 = rearrange(s2, 'b h c -> b c h')
+        mu = self.mu_fc(s2)
+        var = self.var_fc(s2)
+        z = self.reparameterize(mu, var)
+        return mu, var, z
+class RegionEncoder(nn.Module):
+    def __init__(self, N=8):
+        super(RegionEncoder, self).__init__()
+        channels = [8, 16, 32, 32, 64, 64]
+        self.s1 = nn.Conv2d(3, channels[0], kernel_size=3, padding=1, stride=2)
+        self.s2 = nn.Sequential(
+            nn.Conv2d(channels[0], channels[1], kernel_size=3, padding=1, stride=2),
+            nn.BatchNorm2d(channels[1]),
+            nn.LeakyReLU(),
+            nn.Conv2d(channels[1], channels[2], kernel_size=3, padding=1, stride=2),
+            nn.BatchNorm2d(channels[2]),
+            nn.LeakyReLU()
+        )
+        self.heads = nn.ModuleList()
+        for i in range(34):
+            self.heads.append(HeadEncoder(N=N))
+    def forward(self, x, all_mask=None):
+        s1 = self.s1(x)
+        s2 = self.s2(s1)
+        result = []
+        mus = []
+        log_vars = []
+        for i, head in enumerate(self.heads):
+            m = all_mask[:, i, :].unsqueeze(1)
+            mu, var, z = head(s2 * m)
+            result.append(z.unsqueeze(2))
+            mus.append(mu.unsqueeze(2))
+            log_vars.append(var.unsqueeze(2))
+        return torch.cat(mus, dim=2), torch.cat(log_vars, dim=2), torch.cat(result, dim=2)

models/stylegene/util.py ADDED Viewed

	@@ -0,0 +1,30 @@

+import numpy as np
+from PIL import Image
+from torchvision import transforms
+def requires_grad(model, flag=True):
+    for p in model.parameters():
+        p.requires_grad = flag
+def get_keys(d, name):
+    if 'state_dict' in d:
+        d = d['state_dict']
+    d_filt = {k[len(name) + 1:]: v for k, v in d.items() if k[:len(name)] == name}
+    return d_filt
+def load_img(path_img, img_size=(256, 256)):
+    transform = transforms.Compose(
+        [transforms.Resize(img_size),
+         transforms.ToTensor(),
+         transforms.Normalize((0.5, 0.5, 0.5),
+                              (0.5, 0.5, 0.5))])
+    if type(path_img) is np.ndarray:
+        img = Image.fromarray(path_img)
+    else:
+        img = Image.open(path_img).convert('RGB')
+    img = transform(img)
+    img.unsqueeze_(0)
+    return img

preprocess/__init__.py ADDED Viewed

File without changes

preprocess/align_images.py ADDED Viewed

	@@ -0,0 +1,32 @@

+import bz2
+from .face_alignment import image_align
+from .landmarks_detector import LandmarksDetector
+from configs import path_ckpt_landmark68
+LANDMARKS_MODEL_URL = 'http://dlib.net/files/shape_predictor_68_face_landmarks.dat.bz2'
+def unpack_bz2(src_path):
+    data = bz2.BZ2File(src_path).read()
+    dst_path = src_path[:-4]
+    with open(dst_path, 'wb') as fp:
+        fp.write(data)
+    return dst_path
+def init_landmark():
+    landmarks_model_path = unpack_bz2(path_ckpt_landmark68)
+    landmarks_detector = LandmarksDetector(landmarks_model_path)
+    return landmarks_detector
+landmarks_detector = init_landmark()
+def align_face(raw_img, output_size=256):
+    try:
+        face_landmarks = landmarks_detector.get_landmarks(raw_img)[0]
+        aligned_face = image_align(raw_img, face_landmarks, output_size=output_size, transform_size=1024)
+        return aligned_face
+    except:
+        return raw_img

preprocess/face_alignment.py ADDED Viewed

	@@ -0,0 +1,87 @@

+import numpy as np
+import scipy.ndimage
+import os
+import PIL.Image
+def image_align(src, face_landmarks, output_size=1024, transform_size=4096, enable_padding=True, x_scale=1, y_scale=1, em_scale=0.1, alpha=False):
+        # Align function from FFHQ dataset pre-processing step
+        # https://github.com/NVlabs/ffhq-dataset/blob/master/download_ffhq.py
+        lm = np.array(face_landmarks)
+        lm_chin          = lm[0  : 17]  # left-right
+        lm_eyebrow_left  = lm[17 : 22]  # left-right
+        lm_eyebrow_right = lm[22 : 27]  # left-right
+        lm_nose          = lm[27 : 31]  # top-down
+        lm_nostrils      = lm[31 : 36]  # top-down
+        lm_eye_left      = lm[36 : 42]  # left-clockwise
+        lm_eye_right     = lm[42 : 48]  # left-clockwise
+        lm_mouth_outer   = lm[48 : 60]  # left-clockwise
+        lm_mouth_inner   = lm[60 : 68]  # left-clockwise
+        # Calculate auxiliary vectors.
+        eye_left     = np.mean(lm_eye_left, axis=0)
+        eye_right    = np.mean(lm_eye_right, axis=0)
+        eye_avg      = (eye_left + eye_right) * 0.5
+        eye_to_eye   = eye_right - eye_left
+        mouth_left   = lm_mouth_outer[0]
+        mouth_right  = lm_mouth_outer[6]
+        mouth_avg    = (mouth_left + mouth_right) * 0.5
+        eye_to_mouth = mouth_avg - eye_avg
+        # Choose oriented crop rectangle.
+        x = eye_to_eye - np.flipud(eye_to_mouth) * [-1, 1]
+        x /= np.hypot(*x)
+        x *= max(np.hypot(*eye_to_eye) * 2.0, np.hypot(*eye_to_mouth) * 1.8)
+        x *= x_scale
+        y = np.flipud(x) * [-y_scale, y_scale]
+        c = eye_avg + eye_to_mouth * em_scale
+        quad = np.stack([c - x - y, c - x + y, c + x + y, c + x - y])
+        qsize = np.hypot(*x) * 2
+        # Load in-the-wild image.
+        img = PIL.Image.fromarray(src)
+        # Shrink.
+        shrink = int(np.floor(qsize / output_size * 0.5))
+        if shrink > 1:
+            rsize = (int(np.rint(float(img.size[0]) / shrink)), int(np.rint(float(img.size[1]) / shrink)))
+            img = img.resize(rsize, PIL.Image.ANTIALIAS)
+            quad /= shrink
+            qsize /= shrink
+        # Crop.
+        border = max(int(np.rint(qsize * 0.1)), 3)
+        crop = (int(np.floor(min(quad[:,0]))), int(np.floor(min(quad[:,1]))), int(np.ceil(max(quad[:,0]))), int(np.ceil(max(quad[:,1]))))
+        crop = (max(crop[0] - border, 0), max(crop[1] - border, 0), min(crop[2] + border, img.size[0]), min(crop[3] + border, img.size[1]))
+        if crop[2] - crop[0] < img.size[0] or crop[3] - crop[1] < img.size[1]:
+            img = img.crop(crop)
+            quad -= crop[0:2]
+        # Pad.
+        pad = (int(np.floor(min(quad[:,0]))), int(np.floor(min(quad[:,1]))), int(np.ceil(max(quad[:,0]))), int(np.ceil(max(quad[:,1]))))
+        pad = (max(-pad[0] + border, 0), max(-pad[1] + border, 0), max(pad[2] - img.size[0] + border, 0), max(pad[3] - img.size[1] + border, 0))
+        if enable_padding and max(pad) > border - 4:
+            pad = np.maximum(pad, int(np.rint(qsize * 0.3)))
+            img = np.pad(np.float32(img), ((pad[1], pad[3]), (pad[0], pad[2]), (0, 0)), 'reflect')
+            h, w, _ = img.shape
+            y, x, _ = np.ogrid[:h, :w, :1]
+            mask = np.maximum(1.0 - np.minimum(np.float32(x) / pad[0], np.float32(w-1-x) / pad[2]), 1.0 - np.minimum(np.float32(y) / pad[1], np.float32(h-1-y) / pad[3]))
+            blur = qsize * 0.02
+            img += (scipy.ndimage.gaussian_filter(img, [blur, blur, 0]) - img) * np.clip(mask * 3.0 + 1.0, 0.0, 1.0)
+            img += (np.median(img, axis=(0,1)) - img) * np.clip(mask, 0.0, 1.0)
+            img = np.uint8(np.clip(np.rint(img), 0, 255))
+            if alpha:
+                mask = 1-np.clip(3.0 * mask, 0.0, 1.0)
+                mask = np.uint8(np.clip(np.rint(mask*255), 0, 255))
+                img = np.concatenate((img, mask), axis=2)
+                img = PIL.Image.fromarray(img, 'RGBA')
+            else:
+                img = PIL.Image.fromarray(img, 'RGB')
+            quad += pad[:2]
+        # Transform.
+        img = img.transform((transform_size, transform_size), PIL.Image.QUAD, (quad + 0.5).flatten(), PIL.Image.BILINEAR)
+        if output_size < transform_size:
+            img = img.resize((output_size, output_size), PIL.Image.ANTIALIAS)
+        return np.array(img)

preprocess/landmarks_detector.py ADDED Viewed

	@@ -0,0 +1,24 @@

+import dlib
+class LandmarksDetector:
+    def __init__(self, predictor_model_path):
+        """
+        :param predictor_model_path: path to shape_predictor_68_face_landmarks.dat file
+        """
+        self.detector = dlib.get_frontal_face_detector()  # cnn_face_detection_model_v1 also can be used
+        self.shape_predictor = dlib.shape_predictor(predictor_model_path)
+    def get_landmarks(self, img):
+        # img = dlib.load_rgb_image(image) # load from path
+        dets = self.detector(img, 1)
+        all_faces = []
+        for detection in dets:
+            try:
+                face_landmarks = [(item.x, item.y) for item in self.shape_predictor(img, detection).parts()]
+                all_faces.append(face_landmarks)
+            except:
+                print("Exception in get_landmarks()!")
+        return all_faces

requirements.txt ADDED Viewed

	@@ -0,0 +1,19 @@

+Cython==0.29.34
+cytoolz==0.11.0
+dlib==19.24.0
+easydict==1.10
+einops==0.3.2
+gradio==3.32.0
+gradio_client==0.2.5
+huggingface-hub==0.14.1
+omegaconf==2.2.3
+onnx==1.12.0
+onnxruntime==1.9.0
+opencv-contrib-python==4.5.5.64
+opencv-python==4.5.4.60
+opencv-python-headless==4.7.0.72
+pandas==1.4.4
+Pillow==9.2.0
+pytorch-lightning==1.8.1
+torch==1.12.1
+torchvision==0.13.1