Upload 7 files

Browse files

Files changed (7) hide show

.gitignore +145 -0
4x_NMKD-YandereNeoXL_200k.safetensors +3 -0
ESRGAN.py +264 -0
README.md +4 -3
blocks.py +534 -0
requirements.txt +4 -0
upscale.py +71 -0

.gitignore ADDED Viewed

	@@ -0,0 +1,145 @@

+# Byte-compiled / optimized / DLL files
+__pycache__/
+*.py[cod]
+*$py.class
+# C extensions
+*.so
+# Distribution / packaging
+.Python
+build/
+develop-eggs/
+dist/
+downloads/
+eggs/
+.eggs/
+lib/
+lib64/
+parts/
+sdist/
+var/
+wheels/
+share/python-wheels/
+*.egg-info/
+.installed.cfg
+*.egg
+MANIFEST
+# PyInstaller
+#  Usually these files are written by a python script from a template
+#  before PyInstaller builds the exe, so as to inject date/other infos into it.
+*.manifest
+*.spec
+# Installer logs
+pip-log.txt
+pip-delete-this-directory.txt
+# Unit test / coverage reports
+htmlcov/
+.tox/
+.nox/
+.coverage
+.coverage.*
+.cache
+nosetests.xml
+coverage.xml
+*.cover
+*.py,cover
+.hypothesis/
+.pytest_cache/
+cover/
+# Translations
+*.mo
+*.pot
+# Django stuff:
+*.log
+local_settings.py
+db.sqlite3
+db.sqlite3-journal
+# Flask stuff:
+instance/
+.webassets-cache
+# Scrapy stuff:
+.scrapy
+# Sphinx documentation
+docs/_build/
+# PyBuilder
+.pybuilder/
+target/
+# Jupyter Notebook
+.ipynb_checkpoints
+# IPython
+profile_default/
+ipython_config.py
+# pyenv
+#   For a library or package, you might want to ignore these files since the code is
+#   intended to run in multiple environments; otherwise, check them in:
+# .python-version
+# pipenv
+#   According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
+#   However, in case of collaboration, if having platform-specific dependencies or dependencies
+#   having no cross-platform support, pipenv may install dependencies that don't work, or not
+#   install all needed dependencies.
+#Pipfile.lock
+# PEP 582; used by e.g. github.com/David-OConnor/pyflow
+__pypackages__/
+# Celery stuff
+celerybeat-schedule
+celerybeat.pid
+# SageMath parsed files
+*.sage.py
+# Environments
+.env
+.venv
+env/
+venv/
+ENV/
+env.bak/
+venv.bak/
+# Spyder project settings
+.spyderproject
+.spyproject
+# Rope project settings
+.ropeproject
+# mkdocs documentation
+/site
+# mypy
+.mypy_cache/
+.dmypy.json
+dmypy.json
+# Pyre type checker
+.pyre/
+# pytype static type analyzer
+.pytype/
+# Cython debug symbols
+cython_debug/
+# Custom
+*.pth
+input/**/*.*
+output/**/*.*
+.vscode/

4x_NMKD-YandereNeoXL_200k.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f61458c2637415947ee7baf05cdf529d54bb8cd3e36ba47393a3f48a2d1f3d59
+size 66864028

ESRGAN.py ADDED Viewed

	@@ -0,0 +1,264 @@

+import functools, math, re
+from collections import OrderedDict
+import mlx.core as mx
+import mlx.nn as nn
+import numpy as np
+import blocks as B
+from mlx.utils import tree_flatten
+def conv_state_pair_to_mlx(kv):
+    k, v = kv
+    if v.ndim == 4:
+        v = v.transpose(0, 2, 3, 1)
+        v = v.reshape(-1).reshape(v.shape)
+    return re.sub(r'(\.\d+\.)', r'.layers\1', k), v
+# Borrowed from https://github.com/rlaphoenix/VSGAN/blob/master/vsgan/archs/ESRGAN.py
+# Which enhanced stuff that was already here
+class ESRGAN(nn.Module):
+    def __init__(
+        self,
+        state_dict,
+        norm=None,
+        act: str = "leakyrelu",
+        upsampler: str = "upconv",
+        mode: str = "CNA",
+    ) -> None:
+        """
+        ESRGAN - Enhanced Super-Resolution Generative Adversarial Networks.
+        By Xintao Wang, Ke Yu, Shixiang Wu, Jinjin Gu, Yihao Liu, Chao Dong, Yu Qiao,
+        and Chen Change Loy.
+        This is old-arch Residual in Residual Dense Block Network and is not
+        the newest revision that's available at github.com/xinntao/ESRGAN.
+        This is on purpose, the newest Network has severely limited the
+        potential use of the Network with no benefits.
+        This network supports model files from both new and old-arch.
+        Args:
+            norm: Normalization layer
+            act: Activation layer
+            upsampler: Upsample layer. upconv, pixel_shuffle
+            mode: Convolution mode
+        """
+        super().__init__()
+        self._raw_state = state_dict
+        self.norm = norm
+        self.act = act
+        self.upsampler = upsampler
+        self.mode = mode
+        self.state_map = {
+            # currently supports old, new, and newer RRDBNet arch models
+            # ESRGAN, BSRGAN/RealSR, Real-ESRGAN
+            "model.0.weight": ("conv_first.weight",),
+            "model.0.bias": ("conv_first.bias",),
+            "model.1.sub./NB/.weight": ("trunk_conv.weight", "conv_body.weight"),
+            "model.1.sub./NB/.bias": ("trunk_conv.bias", "conv_body.bias"),
+            "model.3.weight": ("upconv1.weight", "conv_up1.weight"),
+            "model.3.bias": ("upconv1.bias", "conv_up1.bias"),
+            "model.6.weight": ("upconv2.weight", "conv_up2.weight"),
+            "model.6.bias": ("upconv2.bias", "conv_up2.bias"),
+            "model.8.weight": ("HRconv.weight", "conv_hr.weight"),
+            "model.8.bias": ("HRconv.bias", "conv_hr.bias"),
+            "model.10.weight": ("conv_last.weight",),
+            "model.10.bias": ("conv_last.bias",),
+            r"model.1.sub.\1.RDB\2.conv\3.0.\4": (
+                r"RRDB_trunk\.(\d+)\.RDB(\d)\.conv(\d+)\.(weight|bias)",
+                r"body\.(\d+)\.rdb(\d)\.conv(\d+)\.(weight|bias)",
+            ),
+        }
+        if "params_ema" in self._raw_state:
+            self._raw_state = self._raw_state["params_ema"]
+        self.num_blocks = self.get_num_blocks()
+        self.plus = any("conv1x1" in k for k in self._raw_state.keys())
+        self._raw_state = self.new_to_old_arch(self._raw_state)
+        self.key_arr = sorted(list(self._raw_state.keys()), key=lambda x: [1 if v == "bias" else 0 if v == "weight" else int(v) if re.match(r'^\d+$', v) else v for v in re.findall(r'[^.]+', x)])
+        # print(self.key_arr)
+        self.in_nc = self._raw_state[self.key_arr[0]].shape[1]
+        self.out_nc = self._raw_state[self.key_arr[-1]].shape[0]
+        self.scale = self.get_scale()
+        self.num_filters = self._raw_state[self.key_arr[0]].shape[0]
+        c2x2 = False
+        if self._raw_state["model.0.weight"].shape[-3] == 2:
+            c2x2 = True
+            self.scale = math.ceil(self.scale ** (1.0 / 3))
+        # Detect if pixelunshuffle was used (Real-ESRGAN)
+        if self.in_nc in (self.out_nc * 4, self.out_nc * 16) and self.out_nc in (
+            self.in_nc / 4,
+            self.in_nc / 16,
+        ):
+            self.shuffle_factor = int(math.sqrt(self.in_nc / self.out_nc))
+        else:
+            self.shuffle_factor = None
+        upsample_block = {
+            "upconv": B.upconv_block,
+            "pixel_shuffle": B.pixelshuffle_block,
+        }.get(self.upsampler)
+        if upsample_block is None:
+            raise NotImplementedError(f"Upsample mode [{self.upsampler}] is not found")
+        if self.scale == 3:
+            upsample_blocks = upsample_block(
+                in_nc=self.num_filters,
+                out_nc=self.num_filters,
+                upscale_factor=3,
+                act_type=self.act,
+                c2x2=c2x2,
+            )
+        else:
+            upsample_blocks = [
+                upsample_block(
+                    in_nc=self.num_filters,
+                    out_nc=self.num_filters,
+                    act_type=self.act,
+                    c2x2=c2x2,
+                )
+                for _ in range(int(math.log(self.scale, 2)))
+            ]
+        self.model = B.sequential(
+            # fea conv
+            B.conv_block(
+                in_nc=self.in_nc,
+                out_nc=self.num_filters,
+                kernel_size=3,
+                norm_type=None,
+                act_type=None,
+                c2x2=c2x2,
+            ),
+            B.ShortcutBlock(
+                B.sequential(
+                    # rrdb blocks
+                    *[
+                        B.RRDB(
+                            nf=self.num_filters,
+                            kernel_size=3,
+                            gc=32,
+                            stride=1,
+                            bias=True,
+                            pad_type="zero",
+                            norm_type=self.norm,
+                            act_type=self.act,
+                            mode="CNA",
+                            plus=self.plus,
+                            c2x2=c2x2,
+                        )
+                        for _ in range(self.num_blocks)
+                    ],
+                    # lr conv
+                    B.conv_block(
+                        in_nc=self.num_filters,
+                        out_nc=self.num_filters,
+                        kernel_size=3,
+                        norm_type=self.norm,
+                        act_type=None,
+                        mode=self.mode,
+                        c2x2=c2x2,
+                    ),
+                )
+            ),
+            *upsample_blocks,
+            # hr_conv0
+            B.conv_block(
+                in_nc=self.num_filters,
+                out_nc=self.num_filters,
+                kernel_size=3,
+                norm_type=None,
+                act_type=self.act,
+                c2x2=c2x2,
+            ),
+            # hr_conv1
+            B.conv_block(
+                in_nc=self.num_filters,
+                out_nc=self.out_nc,
+                kernel_size=3,
+                norm_type=None,
+                act_type=None,
+                c2x2=c2x2,
+            ),
+        )
+        self.load_weights(list(conv_state_pair_to_mlx(p) for p in self._raw_state.items()), strict=True)
+    def new_to_old_arch(self, state):
+        """Convert a new-arch model state dictionary to an old-arch dictionary."""
+        if "params_ema" in state:
+            state = state["params_ema"]
+        if "conv_first.weight" not in state:
+            # model is already old arch, this is a loose check, but should be sufficient
+            return state
+        # add nb to state keys
+        for kind in ("weight", "bias"):
+            self.state_map[f"model.1.sub.{self.num_blocks}.{kind}"] = self.state_map[
+                f"model.1.sub./NB/.{kind}"
+            ]
+            del self.state_map[f"model.1.sub./NB/.{kind}"]
+        old_state = OrderedDict()
+        for old_key, new_keys in self.state_map.items():
+            for new_key in new_keys:
+                if r"\1" in old_key:
+                    for k, v in state.items():
+                        sub = re.sub(new_key, old_key, k)
+                        if sub != k:
+                            old_state[sub] = v
+                else:
+                    if new_key in state:
+                        old_state[old_key] = state[new_key]
+        # Sort by first numeric value of each layer
+        def compare(item1, item2):
+            parts1 = item1.split(".")
+            parts2 = item2.split(".")
+            int1 = int(parts1[1])
+            int2 = int(parts2[1])
+            return int1 - int2
+        sorted_keys = sorted(old_state.keys(), key=functools.cmp_to_key(compare))
+        # Rebuild the output dict in the right order
+        out_dict = OrderedDict((k, old_state[k]) for k in sorted_keys)
+        return out_dict
+    def get_scale(self, min_part: int = 6) -> int:
+        n = 0
+        for part in list(self._raw_state):
+            parts = part.split(".")[1:]
+            if len(parts) == 2:
+                part_num = int(parts[0])
+                if part_num > min_part and parts[1] == "weight":
+                    n += 1
+        return 2**n
+    def get_num_blocks(self) -> int:
+        nbs = []
+        state_keys = self.state_map[r"model.1.sub.\1.RDB\2.conv\3.0.\4"] + (
+            r"model\.\d+\.sub\.(\d+)\.RDB(\d+)\.conv(\d+)\.0\.(weight|bias)",
+        )
+        for state_key in state_keys:
+            for k in self._raw_state:
+                m = re.search(state_key, k)
+                if m:
+                    nbs.append(int(m.group(1)))
+            if nbs:
+                break
+        return max(*nbs) + 1
+    def __call__(self, x):
+        if self.shuffle_factor:
+            x = torch.pixel_unshuffle(x, downscale_factor=self.shuffle_factor)
+        return self.model(x)

README.md CHANGED Viewed

@@ -1,3 +1,4 @@
----
-license: apache-2.0
----

+# kaeru tiny mlx upscaler
+A simple upscale script for `4x_NMKD-YandereNeoXL_200k`
+Based on [joeyballentine/ESRGAN](https://github.com/JoeyBallentine/ESRGAN)

blocks.py ADDED Viewed

	@@ -0,0 +1,534 @@

+#!/usr/bin/env python3
+# -*- coding: utf-8 -*-
+from collections import OrderedDict
+import mlx.core as mx
+import mlx.nn as nn
+####################
+# Basic blocks
+####################
+def act(act_type, inplace=True, neg_slope=0.2, n_prelu=1):
+    # helper selecting activation
+    # neg_slope: for leakyrelu and init of prelu
+    # n_prelu: for p_relu num_parameters
+    act_type = act_type.lower()
+    if act_type == "relu":
+        layer = nn.ReLU()#inplace)
+    elif act_type == "leakyrelu":
+        layer = nn.LeakyReLU(neg_slope)#, inplace)
+    elif act_type == "prelu":
+        layer = nn.PReLU(num_parameters=n_prelu, init=neg_slope)
+    else:
+        raise NotImplementedError(
+            "activation layer [{:s}] is not found".format(act_type)
+        )
+    return layer
+def norm(norm_type, nc):
+    # helper selecting normalization layer
+    norm_type = norm_type.lower()
+    if norm_type == "batch":
+        layer = nn.BatchNorm2d(nc, affine=True)
+    elif norm_type == "instance":
+        layer = nn.InstanceNorm2d(nc, affine=False)
+    else:
+        raise NotImplementedError(
+            "normalization layer [{:s}] is not found".format(norm_type)
+        )
+    return layer
+def pad(pad_type, padding):
+    # helper selecting padding layer
+    # if padding is 'zero', do by conv layers
+    pad_type = pad_type.lower()
+    if padding == 0:
+        return None
+    if pad_type == "reflect":
+        layer = nn.ReflectionPad2d(padding)
+    elif pad_type == "replicate":
+        layer = nn.ReplicationPad2d(padding)
+    else:
+        raise NotImplementedError(
+            "padding layer [{:s}] is not implemented".format(pad_type)
+        )
+    return layer
+def get_valid_padding(kernel_size, dilation):
+    kernel_size = kernel_size + (kernel_size - 1) * (dilation - 1)
+    padding = (kernel_size - 1) // 2
+    return padding
+class ConcatBlock(nn.Module):
+    # Concat the output of a submodule to its input
+    def __init__(self, submodule):
+        super(ConcatBlock, self).__init__()
+        self.sub = submodule
+    def __call__(self, x):
+        output = torch.cat((x, self.sub(x)), dim=1)
+        return output
+    def __repr__(self):
+        tmpstr = "Identity .. \n|"
+        modstr = self.sub.__repr__().replace("\n", "\n|")
+        tmpstr = tmpstr + modstr
+        return tmpstr
+class ShortcutBlock(nn.Module):
+    # Elementwise sum the output of a submodule to its input
+    def __init__(self, submodule):
+        super(ShortcutBlock, self).__init__()
+        self.sub = submodule
+    def __call__(self, x):
+        output = x + self.sub(x)
+        return output
+    def __repr__(self):
+        tmpstr = "Identity + \n|"
+        modstr = self.sub.__repr__().replace("\n", "\n|")
+        tmpstr = tmpstr + modstr
+        return tmpstr
+class ShortcutBlockSPSR(nn.Module):
+    # Elementwise sum the output of a submodule to its input
+    def __init__(self, submodule):
+        super(ShortcutBlockSPSR, self).__init__()
+        self.sub = submodule
+    def __call__(self, x):
+        return x, self.sub
+    def __repr__(self):
+        tmpstr = "Identity + \n|"
+        modstr = self.sub.__repr__().replace("\n", "\n|")
+        tmpstr = tmpstr + modstr
+        return tmpstr
+def sequential(*args):
+    # Flatten Sequential. It unwraps nn.Sequential.
+    if len(args) == 1:
+        if isinstance(args[0], OrderedDict):
+            raise NotImplementedError("sequential does not support OrderedDict input.")
+        return args[0]  # No sequential is needed.
+    modules = []
+    for module in args:
+        if isinstance(module, nn.Sequential):
+            for submodule in module.children()["layers"]:
+                modules.append(submodule)
+        elif isinstance(module, nn.Module):
+            modules.append(module)
+    return nn.Sequential(*modules)
+def conv_block(
+    in_nc,
+    out_nc,
+    kernel_size,
+    stride=1,
+    dilation=1,
+    groups=1,
+    bias=True,
+    pad_type="zero",
+    norm_type=None,
+    act_type="relu",
+    mode="CNA",
+    c2x2=False,
+):
+    """
+    Conv layer with padding, normalization, activation
+    mode: CNA --> Conv -> Norm -> Act
+        NAC --> Norm -> Act --> Conv (Identity Mappings in Deep Residual Networks, ECCV16)
+    """
+    if c2x2:
+        return conv_block_2c2(in_nc, out_nc, act_type=act_type)
+    assert mode in ["CNA", "NAC", "CNAC"], "Wrong conv mode [{:s}]".format(mode)
+    padding = get_valid_padding(kernel_size, dilation)
+    p = pad(pad_type, padding) if pad_type and pad_type != "zero" else None
+    padding = padding if pad_type == "zero" else 0
+    c = nn.Conv2d(
+        in_nc,
+        out_nc,
+        kernel_size=kernel_size,
+        stride=stride,
+        padding=padding,
+        dilation=dilation,
+        bias=bias,
+        **({"groups": groups} if groups != 1 else {}),
+    )
+    a = act(act_type) if act_type else None
+    if "CNA" in mode:
+        n = norm(norm_type, out_nc) if norm_type else None
+        return sequential(p, c, n, a)
+    elif mode == "NAC":
+        if norm_type is None and act_type is not None:
+            a = act(act_type, inplace=False)
+            # Important!
+            # input----ReLU(inplace)----Conv--+----output
+            #        |________________________|
+            # inplace ReLU will modify the input, therefore wrong output
+        n = norm(norm_type, in_nc) if norm_type else None
+        return sequential(n, a, p, c)
+# 2x2x2 Conv Block
+def conv_block_2c2(
+    in_nc,
+    out_nc,
+    act_type="relu",
+):
+    return sequential(
+        nn.Conv2d(in_nc, out_nc, kernel_size=2, padding=1),
+        nn.Conv2d(out_nc, out_nc, kernel_size=2, padding=0),
+        act(act_type) if act_type else None,
+    )
+####################
+# Useful blocks
+####################
+class ResNetBlock(nn.Module):
+    """
+    ResNet Block, 3-3 style
+    with extra residual scaling used in EDSR
+    (Enhanced Deep Residual Networks for Single Image Super-Resolution, CVPRW 17)
+    """
+    def __init__(
+        self,
+        in_nc,
+        mid_nc,
+        out_nc,
+        kernel_size=3,
+        stride=1,
+        dilation=1,
+        groups=1,
+        bias=True,
+        pad_type="zero",
+        norm_type=None,
+        act_type="relu",
+        mode="CNA",
+        res_scale=1,
+    ):
+        super(ResNetBlock, self).__init__()
+        conv0 = conv_block(
+            in_nc,
+            mid_nc,
+            kernel_size,
+            stride,
+            dilation,
+            groups,
+            bias,
+            pad_type,
+            norm_type,
+            act_type,
+            mode,
+        )
+        if mode == "CNA":
+            act_type = None
+        if mode == "CNAC":  # Residual path: |-CNAC-|
+            act_type = None
+            norm_type = None
+        conv1 = conv_block(
+            mid_nc,
+            out_nc,
+            kernel_size,
+            stride,
+            dilation,
+            groups,
+            bias,
+            pad_type,
+            norm_type,
+            act_type,
+            mode,
+        )
+        # if in_nc != out_nc:
+        #     self.project = conv_block(in_nc, out_nc, 1, stride, dilation, 1, bias, pad_type, \
+        #         None, None)
+        #     print('Need a projecter in ResNetBlock.')
+        # else:
+        #     self.project = lambda x:x
+        self.res = sequential(conv0, conv1)
+        self.res_scale = res_scale
+    def __call__(self, x):
+        res = self.res(x).mul(self.res_scale)
+        return x + res
+class RRDB(nn.Module):
+    """
+    Residual in Residual Dense Block
+    (ESRGAN: Enhanced Super-Resolution Generative Adversarial Networks)
+    """
+    def __init__(
+        self,
+        nf,
+        kernel_size=3,
+        gc=32,
+        stride=1,
+        bias=1,
+        pad_type="zero",
+        norm_type=None,
+        act_type="leakyrelu",
+        mode="CNA",
+        convtype="Conv2D",
+        spectral_norm=False,
+        plus=False,
+        c2x2=False,
+    ):
+        super(RRDB, self).__init__()
+        self.RDB1 = ResidualDenseBlock_5C(
+            nf,
+            kernel_size,
+            gc,
+            stride,
+            bias,
+            pad_type,
+            norm_type,
+            act_type,
+            mode,
+            plus=plus,
+            c2x2=c2x2,
+        )
+        self.RDB2 = ResidualDenseBlock_5C(
+            nf,
+            kernel_size,
+            gc,
+            stride,
+            bias,
+            pad_type,
+            norm_type,
+            act_type,
+            mode,
+            plus=plus,
+            c2x2=c2x2,
+        )
+        self.RDB3 = ResidualDenseBlock_5C(
+            nf,
+            kernel_size,
+            gc,
+            stride,
+            bias,
+            pad_type,
+            norm_type,
+            act_type,
+            mode,
+            plus=plus,
+            c2x2=c2x2,
+        )
+    def __call__(self, x):
+        out = self.RDB1(x)
+        out = self.RDB2(out)
+        out = self.RDB3(out)
+        return out * 0.2 + x
+class ResidualDenseBlock_5C(nn.Module):
+    """
+    Residual Dense Block
+    style: 5 convs
+    The core module of paper: (Residual Dense Network for Image Super-Resolution, CVPR 18)
+    Modified options that can be used:
+        - "Partial Convolution based Padding" arXiv:1811.11718
+        - "Spectral normalization" arXiv:1802.05957
+        - "ICASSP 2020 - ESRGAN+ : Further Improving ESRGAN" N. C.
+            {Rakotonirina} and A. {Rasoanaivo}
+    Args:
+        nf (int): Channel number of intermediate features (num_feat).
+        gc (int): Channels for each growth (num_grow_ch: growth channel,
+            i.e. intermediate channels).
+        convtype (str): the type of convolution to use. Default: 'Conv2D'
+        gaussian_noise (bool): enable the ESRGAN+ gaussian noise (no new
+            trainable parameters)
+        plus (bool): enable the additional residual paths from ESRGAN+
+            (adds trainable parameters)
+    """
+    def __init__(
+        self,
+        nf=64,
+        kernel_size=3,
+        gc=32,
+        stride=1,
+        bias=1,
+        pad_type="zero",
+        norm_type=None,
+        act_type="leakyrelu",
+        mode="CNA",
+        plus=False,
+        c2x2=False,
+    ):
+        super(ResidualDenseBlock_5C, self).__init__()
+        ## +
+        self.conv1x1 = conv1x1(nf, gc) if plus else None
+        ## +
+        self.conv1 = conv_block(
+            nf,
+            gc,
+            kernel_size,
+            stride,
+            bias=bias,
+            pad_type=pad_type,
+            norm_type=norm_type,
+            act_type=act_type,
+            mode=mode,
+            c2x2=c2x2,
+        )
+        self.conv2 = conv_block(
+            nf + gc,
+            gc,
+            kernel_size,
+            stride,
+            bias=bias,
+            pad_type=pad_type,
+            norm_type=norm_type,
+            act_type=act_type,
+            mode=mode,
+            c2x2=c2x2,
+        )
+        self.conv3 = conv_block(
+            nf + 2 * gc,
+            gc,
+            kernel_size,
+            stride,
+            bias=bias,
+            pad_type=pad_type,
+            norm_type=norm_type,
+            act_type=act_type,
+            mode=mode,
+            c2x2=c2x2,
+        )
+        self.conv4 = conv_block(
+            nf + 3 * gc,
+            gc,
+            kernel_size,
+            stride,
+            bias=bias,
+            pad_type=pad_type,
+            norm_type=norm_type,
+            act_type=act_type,
+            mode=mode,
+            c2x2=c2x2,
+        )
+        if mode == "CNA":
+            last_act = None
+        else:
+            last_act = act_type
+        self.conv5 = conv_block(
+            nf + 4 * gc,
+            nf,
+            3,
+            stride,
+            bias=bias,
+            pad_type=pad_type,
+            norm_type=norm_type,
+            act_type=last_act,
+            mode=mode,
+            c2x2=c2x2,
+        )
+    def __call__(self, x):
+        x1 = self.conv1(x)
+        x2 = self.conv2(mx.concatenate((x, x1), axis=3))
+        if self.conv1x1:
+            x2 = x2 + self.conv1x1(x)  # +
+        x3 = self.conv3(mx.concatenate((x, x1, x2), axis=3))
+        x4 = self.conv4(mx.concatenate((x, x1, x2, x3), axis=3))
+        if self.conv1x1:
+            x4 = x4 + x2  # +
+        x5 = self.conv5(mx.concatenate((x, x1, x2, x3, x4), axis=3))
+        return x5 * 0.2 + x
+def conv1x1(in_planes, out_planes, stride=1):
+    return nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride, bias=False)
+####################
+# Upsampler
+####################
+def pixelshuffle_block(
+    in_nc,
+    out_nc,
+    upscale_factor=2,
+    kernel_size=3,
+    stride=1,
+    bias=True,
+    pad_type="zero",
+    norm_type=None,
+    act_type="relu",
+):
+    """
+    Pixel shuffle layer
+    (Real-Time Single Image and Video Super-Resolution Using an Efficient Sub-Pixel Convolutional
+    Neural Network, CVPR17)
+    """
+    conv = conv_block(
+        in_nc,
+        out_nc * (upscale_factor**2),
+        kernel_size,
+        stride,
+        bias=bias,
+        pad_type=pad_type,
+        norm_type=None,
+        act_type=None,
+    )
+    pixel_shuffle = nn.PixelShuffle(upscale_factor)
+    n = norm(norm_type, out_nc) if norm_type else None
+    a = act(act_type) if act_type else None
+    return sequential(conv, pixel_shuffle, n, a)
+def upconv_block(
+    in_nc,
+    out_nc,
+    upscale_factor=2,
+    kernel_size=3,
+    stride=1,
+    bias=True,
+    pad_type="zero",
+    norm_type=None,
+    act_type="relu",
+    mode="nearest",
+    c2x2=False,
+):
+    # Up conv
+    # described in https://distill.pub/2016/deconv-checkerboard/
+    upsample = nn.Upsample(scale_factor=upscale_factor, mode=mode)
+    conv = conv_block(
+        in_nc,
+        out_nc,
+        kernel_size,
+        stride,
+        bias=bias,
+        pad_type=pad_type,
+        norm_type=norm_type,
+        act_type=act_type,
+        c2x2=c2x2,
+    )
+    return sequential(upsample, conv)

requirements.txt ADDED Viewed

	@@ -0,0 +1,4 @@

+mlx==0.20.0
+numpy==2.1.3
+pillow==11.0.0
+tqdm==4.67.0

upscale.py ADDED Viewed

	@@ -0,0 +1,71 @@

+import os, re, argparse, threading
+import mlx.core as mx
+import numpy as np
+from PIL import Image, PngImagePlugin
+from tqdm import tqdm
+from ESRGAN import ESRGAN
+def parse_args():
+    parser = argparse.ArgumentParser(description="Process tile size, padding, and file paths.")
+    parser.add_argument('--model', metavar='file_path', type=str, default='4x_NMKD-YandereNeoXL_200k.safetensors', help='Path to the model file')
+    parser.add_argument('--tile_size', metavar='256', type=int, default=256, help='Size of each tile (default: 256)')
+    parser.add_argument('--tile_pad', metavar='10', type=int, default=10, help='Padding around each tile (default: 10)')
+    parser.add_argument('files', metavar='in_file_path', type=str, nargs='+', help='List of file paths to process')
+    return parser.parse_args()
+def load_model(model_path):
+    model = ESRGAN(mx.load(model_path))
+    return mx.compile(model), model.scale
+def upscale_img(args, model, file_path, scale=4.0):
+    ts, tp, s = (args.tile_size, args.tile_pad, scale)
+    img_in = Image.open(file_path)
+    png_info = PngImagePlugin.PngInfo()
+    for k, v in (getattr(img_in, "text", None) or {}).items():
+        png_info.add_text(k, v)
+    img_save_argv = {
+        "icc_profile": img_in.info.get('icc_profile'),
+        "pnginfo": png_info,
+    }
+    img_in = mx.array(np.array(img_in.convert("RGB"), dtype=np.float32))[None] / 255.0
+    _, H, W, C = img_in.shape
+    mx.eval(img_in)
+    img_out = mx.zeros((1, H*s, W*s, C), dtype=mx.uint8)
+    mx.eval(img_out)
+    for hi, wj in tqdm([(hi, wj) for hi in range(0, H, ts) for wj in range(0, W, ts)]):
+        phs = min(hi, tp)
+        pws = min(wj, tp)
+        img_out[:, hi*4:(hi+ts)*s, wj*4:(wj+ts)*s, :] = (
+            model(img_in[:, max(0,hi-tp):hi+ts+tp, max(0, wj-tp):wj+ts+tp, :])[:, phs*s:(ts+phs)*s, pws*s:(ts+pws)*s, :] * 255.0
+        ).astype(mx.uint8)
+        mx.eval(img_out)
+    img_out = np.array(img_out[0], copy=False)
+    img_out = Image.fromarray(img_out)
+    img_out.save(re.sub(r'(\.\w+)$', r'_4x.png', file_path), **img_save_argv)
+def main():
+    print("\033[1;32mkaeru tiny mlx upscaler v0.1\033[0m")
+    mx.metal.set_cache_limit(0)
+    args = parse_args()
+    model, scale = load_model(args.model)
+    for file_path in args.files:
+        print(f"Upscaling {file_path}")
+        upscale_img(args, model, file_path, scale=scale)
+if __name__ == "__main__":
+    th = threading.Thread(target=main)
+    th.start()
+    th.join()