Spaces:

Fabrice-TIERCELIN
/

for-pinokio

Runtime error

App Files Files Community

Fabrice-TIERCELIN commited on Aug 7, 2024

Commit

4cae45e

verified ·

1 Parent(s): 7f6739c

Add working files

Browse files

Files changed (6) hide show

README.md +7 -12
app.py +110 -273
briarmbg.py +455 -0
foo.py +2 -0
input.jpg +0 -0
requirements.txt +9 -26

README.md CHANGED Viewed

@@ -1,18 +1,13 @@
 ---
-title: Text-to-Audio
-emoji: 🔊
-colorFrom: gray
-colorTo: gray
-tags:
-- sound generation
-- language models
-- LLMs
 sdk: gradio
-sdk_version: 4.40.0
 app_file: app.py
 pinned: false
-license: openrail
-short_description: Sound effect from description
 ---
-Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

 ---
+title: BRIA RMBG 1.4
+emoji: 💻
+colorFrom: red
+colorTo: red
 sdk: gradio
+sdk_version: 4.16.0
 app_file: app.py
 pinned: false
+license: other
 ---
+Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

app.py CHANGED Viewed

@@ -1,278 +1,115 @@
-import gradio as gr
-import json
 import torch
-import time
-import random
-try:
-    # Only on HuggingFace
-    import spaces
-    is_space_imported = True
-except ImportError:
-    is_space_imported = False
-from tqdm import tqdm
-from huggingface_hub import snapshot_download
-from models import AudioDiffusion, DDPMScheduler
-from audioldm.audio.stft import TacotronSTFT
-from audioldm.variational_autoencoder import AutoencoderKL
-from pydub import AudioSegment
-max_64_bit_int = 2**63 - 1
-# Automatic device detection
 if torch.cuda.is_available():
-    device_type = "cuda"
-    device_selection = "cuda:0"
 else:
-    device_type = "cpu"
-    device_selection = "cpu"
-class Tango:
-    def __init__(self, name = "declare-lab/tango2", device = device_selection):
-        path = snapshot_download(repo_id = name)
-        vae_config = json.load(open("{}/vae_config.json".format(path)))
-        stft_config = json.load(open("{}/stft_config.json".format(path)))
-        main_config = json.load(open("{}/main_config.json".format(path)))
-        self.vae = AutoencoderKL(**vae_config).to(device)
-        self.stft = TacotronSTFT(**stft_config).to(device)
-        self.model = AudioDiffusion(**main_config).to(device)
-        vae_weights = torch.load("{}/pytorch_model_vae.bin".format(path), map_location = device)
-        stft_weights = torch.load("{}/pytorch_model_stft.bin".format(path), map_location = device)
-        main_weights = torch.load("{}/pytorch_model_main.bin".format(path), map_location = device)
-        self.vae.load_state_dict(vae_weights)
-        self.stft.load_state_dict(stft_weights)
-        self.model.load_state_dict(main_weights)
-        print ("Successfully loaded checkpoint from:", name)
-        self.vae.eval()
-        self.stft.eval()
-        self.model.eval()
-        self.scheduler = DDPMScheduler.from_pretrained(main_config["scheduler_name"], subfolder = "scheduler")
-    def chunks(self, lst, n):
-        # Yield successive n-sized chunks from a list
-        for i in range(0, len(lst), n):
-            yield lst[i:i + n]
-    def generate(self, prompt, steps = 100, guidance = 3, samples = 1, disable_progress = True):
-        # Generate audio for a single prompt string
-        with torch.no_grad():
-            latents = self.model.inference([prompt], self.scheduler, steps, guidance, samples, disable_progress = disable_progress)
-            mel = self.vae.decode_first_stage(latents)
-            wave = self.vae.decode_to_waveform(mel)
-        return wave
-    def generate_for_batch(self, prompts, steps = 200, guidance = 3, samples = 1, batch_size = 8, disable_progress = True):
-        # Generate audio for a list of prompt strings
-        outputs = []
-        for k in tqdm(range(0, len(prompts), batch_size)):
-            batch = prompts[k: k + batch_size]
-            with torch.no_grad():
-                latents = self.model.inference(batch, self.scheduler, steps, guidance, samples, disable_progress = disable_progress)
-                mel = self.vae.decode_first_stage(latents)
-                wave = self.vae.decode_to_waveform(mel)
-                outputs += [item for item in wave]
-        if samples == 1:
-            return outputs
-        return list(self.chunks(outputs, samples))
-# Initialize TANGO
-tango = Tango(device = "cpu")
-tango.vae.to(device_type)
-tango.stft.to(device_type)
-tango.model.to(device_type)
-def update_seed(is_randomize_seed, seed):
-    if is_randomize_seed:
-        return random.randint(0, max_64_bit_int)
-    return seed
-def check(
-    prompt,
-    output_number,
-    steps,
-    guidance,
-    is_randomize_seed,
-    seed
-):
-    if prompt is None or prompt == "":
-        raise gr.Error("Please provide a prompt input.")
-    if not output_number in [1, 2, 3]:
-        raise gr.Error("Please ask for 1, 2 or 3 output files.")
-def update_output(output_format, output_number):
-    return [
-        gr.update(format = output_format),
-        gr.update(format = output_format, visible = (2 <= output_number)),
-        gr.update(format = output_format, visible = (output_number == 3)),
-        gr.update(visible = False)
-    ]
-def text2audio(
-    prompt,
-    output_number,
-    steps,
-    guidance,
-    is_randomize_seed,
-    seed
-):
-    start = time.time()
-    if seed is None:
-        seed = random.randint(0, max_64_bit_int)
-    random.seed(seed)
-    torch.manual_seed(seed)
-    output_wave = tango.generate(prompt, steps, guidance, output_number)
-    output_wave_1 = gr.make_waveform((16000, output_wave[0]))
-    output_wave_2 = gr.make_waveform((16000, output_wave[1])) if (2 <= output_number) else None
-    output_wave_3 = gr.make_waveform((16000, output_wave[2])) if (output_number == 3) else None
-    end = time.time()
-    secondes = int(end - start)
-    minutes = secondes // 60
-    secondes = secondes - (minutes * 60)
-    hours = minutes // 60
-    minutes = minutes - (hours * 60)
-    return [
-        output_wave_1,
-        output_wave_2,
-        output_wave_3,
-        gr.update(visible = True, value = "Start again to get a different result. The output have been generated in " + ((str(hours) + " h, ") if hours != 0 else "") + ((str(minutes) + " min, ") if hours != 0 or minutes != 0 else "") + str(secondes) + " sec.")
-    ]
-if is_space_imported:
-    text2audio = spaces.GPU(text2audio, duration = 420)
-# Gradio interface
-with gr.Blocks() as interface:
-    gr.Markdown("""
-        <p style="text-align: center;">
-        <b><big><big><big>Text-to-Audio</big></big></big></b>
-        <br/>Generates 10 seconds of sound effects from description, freely, without account, without watermark
-        </p>
-        <br/>
-        <br/>
-        ✨ Powered by <i>Tango 2</i> AI.
-        <br/>
-        <ul>
-        <li>If you need <b>47 seconds</b> of audio, I recommend to use <i>Stable Audio</i>,</li>
-        <li>If you need to generate <b>music</b>, I recommend to use <i>MusicGen</i>,</li>
-        </ul>
-        <br/>
-        """ + ("🏃‍♀️ Estimated time: few minutes. Current device: GPU." if torch.cuda.is_available() else "🐌 Slow process... ~5 min. Current device: CPU.") + """
-        Your computer must <b><u>not</u></b> enter into standby mode.<br/>You can duplicate this space on a free account, it's designed to work on CPU, GPU and ZeroGPU.<br/>
-        <a href='https://huggingface.co/spaces/Fabrice-TIERCELIN/Text-to-Audio?duplicate=true&hidden=public&hidden=public'><img src='https://img.shields.io/badge/-Duplicate%20Space-blue?labelColor=white&style=flat&logo=data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAYAAAAf8/9hAAAAAXNSR0IArs4c6QAAAP5JREFUOE+lk7FqAkEURY+ltunEgFXS2sZGIbXfEPdLlnxJyDdYB62sbbUKpLbVNhyYFzbrrA74YJlh9r079973psed0cvUD4A+4HoCjsA85X0Dfn/RBLBgBDxnQPfAEJgBY+A9gALA4tcbamSzS4xq4FOQAJgCDwV2CPKV8tZAJcAjMMkUe1vX+U+SMhfAJEHasQIWmXNN3abzDwHUrgcRGmYcgKe0bxrblHEB4E/pndMazNpSZGcsZdBlYJcEL9Afo75molJyM2FxmPgmgPqlWNLGfwZGG6UiyEvLzHYDmoPkDDiNm9JR9uboiONcBXrpY1qmgs21x1QwyZcpvxt9NS09PlsPAAAAAElFTkSuQmCC&logoWidth=14'></a>
-        <br/>
-        ⚖️ You can use, modify and share the generated sounds but not for commercial uses.
-        """
-    )
-    input_text = gr.Textbox(label = "Prompt", value = "Snort of a horse", lines = 2, autofocus = True)
-    with gr.Accordion("Advanced options", open = False):
-        output_format = gr.Radio(label = "Output format", info = "The file you can dowload", choices = ["mp3", "wav"], value = "wav")
-        output_number = gr.Slider(label = "Number of generations", info = "1, 2 or 3 output files", minimum = 1, maximum = 3, value = 1, step = 1, interactive = True)
-        denoising_steps = gr.Slider(label = "Steps", info = "lower=faster & variant, higher=audio quality & similar", minimum = 10, maximum = 200, value = 10, step = 1, interactive = True)
-        guidance_scale = gr.Slider(label = "Guidance Scale", info = "lower=audio quality, higher=follow the prompt", minimum = 1, maximum = 10, value = 3, step = 0.1, interactive = True)
-        randomize_seed = gr.Checkbox(label = "\U0001F3B2 Randomize seed", value = True, info = "If checked, result is always different")
-        seed = gr.Slider(minimum = 0, maximum = max_64_bit_int, step = 1, randomize = True, label = "Seed")
-    submit = gr.Button("🚀 Generate", variant = "primary")
-    output_audio_1 = gr.Audio(label = "Generated Audio #1/3", format = "wav", type="numpy", autoplay = True)
-    output_audio_2 = gr.Audio(label = "Generated Audio #2/3", format = "wav", type="numpy")
-    output_audio_3 = gr.Audio(label = "Generated Audio #3/3", format = "wav", type="numpy")
-    information = gr.Label(label = "Information")
-    submit.click(fn = update_seed, inputs = [
-        randomize_seed,
-        seed
-    ], outputs = [
-        seed
-    ], queue = False, show_progress = False).then(fn = check, inputs = [
-        input_text,
-        output_number,
-        denoising_steps,
-        guidance_scale,
-        randomize_seed,
-        seed
-    ], outputs = [], queue = False, show_progress = False).success(fn = update_output, inputs = [
-        output_format,
-        output_number
-    ], outputs = [
-        output_audio_1,
-        output_audio_2,
-        output_audio_3,
-        information
-    ], queue = False, show_progress = False).success(fn = text2audio, inputs = [
-        input_text,
-        output_number,
-        denoising_steps,
-        guidance_scale,
-        randomize_seed,
-        seed
-    ], outputs = [
-        output_audio_1,
-        output_audio_2,
-        output_audio_3,
-        information
-    ], scroll_to_output = True)
-    gr.Examples(
-        fn = text2audio,
-	    inputs = [
-            input_text,
-            output_number,
-            denoising_steps,
-            guidance_scale,
-            randomize_seed,
-            seed
-        ],
-	    outputs = [
-            output_audio_1,
-            output_audio_2,
-            output_audio_3,
-            information
-        ],
-        examples = [
-                ["A hammer is hitting a wooden surface", 3, 100, 3, False, 123],
-                ["Peaceful and calming ambient music with singing bowl and other instruments.", 3, 100, 3, False, 123],
-                ["A man is speaking in a small room.", 2, 100, 3, False, 123],
-                ["A female is speaking followed by footstep sound", 1, 100, 3, False, 123],
-                ["Wooden table tapping sound followed by water pouring sound.", 3, 200, 3, False, 123],
-            ],
-        cache_examples = "lazy" if is_space_imported else False,
-    )
-    gr.Markdown(
-        """
-        ## How to prompt your sound
-        You can use round brackets to increase the importance of a part:
-        ```
-        Peaceful and (calming) ambient music with singing bowl and other instruments
-        ```
-        You can use several levels of round brackets to even more increase the importance of a part:
-        ```
-        (Peaceful) and ((calming)) ambient music with singing bowl and other instruments
-        ```
-        You can use number instead of several round brackets:
-        ```
-        (Peaceful:1.5) and ((calming)) ambient music with singing bowl and other instruments
-        ```
-        You can do the same thing with square brackets to decrease the importance of a part:
-        ```
-        (Peaceful:1.5) and ((calming)) ambient music with [singing:2] bowl and other instruments
-        """
-    )
-    if __name__ == "__main__":
-        interface.launch(share = False)

+import numpy as np
 import torch
+import torch.nn.functional as F
+from torchvision.transforms.functional import normalize
+from huggingface_hub import hf_hub_download
+import gradio as gr
+from gradio_imageslider import ImageSlider
+from briarmbg import BriaRMBG
+import PIL
+from PIL import Image
+from typing import Tuple
+net=BriaRMBG()
+# model_path = "./model1.pth"
+#model_path = hf_hub_download("briaai/RMBG-1.4", 'model.pth')
+model_path = hf_hub_download("cocktailpeanut/gbmr", 'model.pth')
 if torch.cuda.is_available():
+    net.load_state_dict(torch.load(model_path))
+    net=net.cuda()
+    device = "cuda"
+elif torch.backends.mps.is_available():
+    net.load_state_dict(torch.load(model_path,map_location="mps"))
+    net=net.to("mps")
+    device = "mps"
 else:
+    net.load_state_dict(torch.load(model_path,map_location="cpu"))
+    device = "cpu"
+net.eval()
+def resize_image(image):
+    image = image.convert('RGB')
+    model_input_size = (1024, 1024)
+    image = image.resize(model_input_size, Image.BILINEAR)
+    return image
+def process(image):
+    # prepare input
+    orig_image = Image.fromarray(image)
+    w,h = orig_im_size = orig_image.size
+    image = resize_image(orig_image)
+    im_np = np.array(image)
+    im_tensor = torch.tensor(im_np, dtype=torch.float32).permute(2,0,1)
+    im_tensor = torch.unsqueeze(im_tensor,0)
+    im_tensor = torch.divide(im_tensor,255.0)
+    im_tensor = normalize(im_tensor,[0.5,0.5,0.5],[1.0,1.0,1.0])
+    if device == "cuda":
+        im_tensor=im_tensor.cuda()
+    elif device == "mps":
+        im_tensor=im_tensor.to("mps")
+    #inference
+    result=net(im_tensor)
+    # post process
+    result = torch.squeeze(F.interpolate(result[0][0], size=(h,w), mode='bilinear') ,0)
+    ma = torch.max(result)
+    mi = torch.min(result)
+    result = (result-mi)/(ma-mi)
+    # image to pil
+    im_array = (result*255).cpu().data.numpy().astype(np.uint8)
+    pil_im = Image.fromarray(np.squeeze(im_array))
+    # paste the mask on the original image
+    new_im = Image.new("RGBA", pil_im.size, (0,0,0,0))
+    new_im.paste(orig_image, mask=pil_im)
+    # new_orig_image = orig_image.convert('RGBA')
+    return new_im
+    # return [new_orig_image, new_im]
+# block = gr.Blocks().queue()
+# with block:
+#     gr.Markdown("## BRIA RMBG 1.4")
+#     gr.HTML('''
+#       <p style="margin-bottom: 10px; font-size: 94%">
+#         This is a demo for BRIA RMBG 1.4 that using
+#         <a href="https://huggingface.co/briaai/RMBG-1.4" target="_blank">BRIA RMBG-1.4 image matting model</a> as backbone.
+#       </p>
+#     ''')
+#     with gr.Row():
+#         with gr.Column():
+#             input_image = gr.Image(sources=None, type="pil") # None for upload, ctrl+v and webcam
+#             # input_image = gr.Image(sources=None, type="numpy") # None for upload, ctrl+v and webcam
+#             run_button = gr.Button(value="Run")
+#         with gr.Column():
+#             result_gallery = gr.Gallery(label='Output', show_label=False, elem_id="gallery", columns=[1], height='auto')
+#     ips = [input_image]
+#     run_button.click(fn=process, inputs=ips, outputs=[result_gallery])
+# block.launch(debug = True)
+# block = gr.Blocks().queue()
+gr.Markdown("## BRIA RMBG 1.4")
+gr.HTML('''
+  <p style="margin-bottom: 10px; font-size: 94%">
+    This is a demo for BRIA RMBG 1.4 that using
+    <a href="https://huggingface.co/briaai/RMBG-1.4" target="_blank">BRIA RMBG-1.4 image matting model</a> as backbone.
+  </p>
+''')
+title = "Background Removal"
+description = r"""Background removal model developed by <a href='https://BRIA.AI' target='_blank'><b>BRIA.AI</b></a>, trained on a carefully selected dataset and is available as an open-source model for non-commercial use.<br>
+For test upload your image and wait. Read more at model card <a href='https://huggingface.co/briaai/RMBG-1.4' target='_blank'><b>briaai/RMBG-1.4</b></a>.<br>
+"""
+examples = [['./input.jpg'],]
+# output = ImageSlider(position=0.5,label='Image without background', type="pil", show_download_button=True)
+# demo = gr.Interface(fn=process,inputs="image", outputs=output, examples=examples, title=title, description=description)
+demo = gr.Interface(fn=process,inputs="image", outputs="image", examples=examples, title=title, description=description)
+if __name__ == "__main__":
+    demo.launch(share=False)

briarmbg.py ADDED Viewed

	@@ -0,0 +1,455 @@

+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+class REBNCONV(nn.Module):
+    def __init__(self,in_ch=3,out_ch=3,dirate=1,stride=1):
+        super(REBNCONV,self).__init__()
+        self.conv_s1 = nn.Conv2d(in_ch,out_ch,3,padding=1*dirate,dilation=1*dirate,stride=stride)
+        self.bn_s1 = nn.BatchNorm2d(out_ch)
+        self.relu_s1 = nn.ReLU(inplace=True)
+    def forward(self,x):
+        hx = x
+        xout = self.relu_s1(self.bn_s1(self.conv_s1(hx)))
+        return xout
+## upsample tensor 'src' to have the same spatial size with tensor 'tar'
+def _upsample_like(src,tar):
+    src = F.interpolate(src,size=tar.shape[2:],mode='bilinear')
+    return src
+### RSU-7 ###
+class RSU7(nn.Module):
+    def __init__(self, in_ch=3, mid_ch=12, out_ch=3, img_size=512):
+        super(RSU7,self).__init__()
+        self.in_ch = in_ch
+        self.mid_ch = mid_ch
+        self.out_ch = out_ch
+        self.rebnconvin = REBNCONV(in_ch,out_ch,dirate=1) ## 1 -> 1/2
+        self.rebnconv1 = REBNCONV(out_ch,mid_ch,dirate=1)
+        self.pool1 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv2 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool2 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv3 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool3 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv4 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool4 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv5 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool5 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv6 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.rebnconv7 = REBNCONV(mid_ch,mid_ch,dirate=2)
+        self.rebnconv6d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv5d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv4d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv3d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv2d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv1d = REBNCONV(mid_ch*2,out_ch,dirate=1)
+    def forward(self,x):
+        b, c, h, w = x.shape
+        hx = x
+        hxin = self.rebnconvin(hx)
+        hx1 = self.rebnconv1(hxin)
+        hx = self.pool1(hx1)
+        hx2 = self.rebnconv2(hx)
+        hx = self.pool2(hx2)
+        hx3 = self.rebnconv3(hx)
+        hx = self.pool3(hx3)
+        hx4 = self.rebnconv4(hx)
+        hx = self.pool4(hx4)
+        hx5 = self.rebnconv5(hx)
+        hx = self.pool5(hx5)
+        hx6 = self.rebnconv6(hx)
+        hx7 = self.rebnconv7(hx6)
+        hx6d =  self.rebnconv6d(torch.cat((hx7,hx6),1))
+        hx6dup = _upsample_like(hx6d,hx5)
+        hx5d =  self.rebnconv5d(torch.cat((hx6dup,hx5),1))
+        hx5dup = _upsample_like(hx5d,hx4)
+        hx4d = self.rebnconv4d(torch.cat((hx5dup,hx4),1))
+        hx4dup = _upsample_like(hx4d,hx3)
+        hx3d = self.rebnconv3d(torch.cat((hx4dup,hx3),1))
+        hx3dup = _upsample_like(hx3d,hx2)
+        hx2d = self.rebnconv2d(torch.cat((hx3dup,hx2),1))
+        hx2dup = _upsample_like(hx2d,hx1)
+        hx1d = self.rebnconv1d(torch.cat((hx2dup,hx1),1))
+        return hx1d + hxin
+### RSU-6 ###
+class RSU6(nn.Module):
+    def __init__(self, in_ch=3, mid_ch=12, out_ch=3):
+        super(RSU6,self).__init__()
+        self.rebnconvin = REBNCONV(in_ch,out_ch,dirate=1)
+        self.rebnconv1 = REBNCONV(out_ch,mid_ch,dirate=1)
+        self.pool1 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv2 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool2 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv3 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool3 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv4 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool4 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv5 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.rebnconv6 = REBNCONV(mid_ch,mid_ch,dirate=2)
+        self.rebnconv5d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv4d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv3d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv2d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv1d = REBNCONV(mid_ch*2,out_ch,dirate=1)
+    def forward(self,x):
+        hx = x
+        hxin = self.rebnconvin(hx)
+        hx1 = self.rebnconv1(hxin)
+        hx = self.pool1(hx1)
+        hx2 = self.rebnconv2(hx)
+        hx = self.pool2(hx2)
+        hx3 = self.rebnconv3(hx)
+        hx = self.pool3(hx3)
+        hx4 = self.rebnconv4(hx)
+        hx = self.pool4(hx4)
+        hx5 = self.rebnconv5(hx)
+        hx6 = self.rebnconv6(hx5)
+        hx5d =  self.rebnconv5d(torch.cat((hx6,hx5),1))
+        hx5dup = _upsample_like(hx5d,hx4)
+        hx4d = self.rebnconv4d(torch.cat((hx5dup,hx4),1))
+        hx4dup = _upsample_like(hx4d,hx3)
+        hx3d = self.rebnconv3d(torch.cat((hx4dup,hx3),1))
+        hx3dup = _upsample_like(hx3d,hx2)
+        hx2d = self.rebnconv2d(torch.cat((hx3dup,hx2),1))
+        hx2dup = _upsample_like(hx2d,hx1)
+        hx1d = self.rebnconv1d(torch.cat((hx2dup,hx1),1))
+        return hx1d + hxin
+### RSU-5 ###
+class RSU5(nn.Module):
+    def __init__(self, in_ch=3, mid_ch=12, out_ch=3):
+        super(RSU5,self).__init__()
+        self.rebnconvin = REBNCONV(in_ch,out_ch,dirate=1)
+        self.rebnconv1 = REBNCONV(out_ch,mid_ch,dirate=1)
+        self.pool1 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv2 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool2 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv3 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool3 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv4 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.rebnconv5 = REBNCONV(mid_ch,mid_ch,dirate=2)
+        self.rebnconv4d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv3d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv2d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv1d = REBNCONV(mid_ch*2,out_ch,dirate=1)
+    def forward(self,x):
+        hx = x
+        hxin = self.rebnconvin(hx)
+        hx1 = self.rebnconv1(hxin)
+        hx = self.pool1(hx1)
+        hx2 = self.rebnconv2(hx)
+        hx = self.pool2(hx2)
+        hx3 = self.rebnconv3(hx)
+        hx = self.pool3(hx3)
+        hx4 = self.rebnconv4(hx)
+        hx5 = self.rebnconv5(hx4)
+        hx4d = self.rebnconv4d(torch.cat((hx5,hx4),1))
+        hx4dup = _upsample_like(hx4d,hx3)
+        hx3d = self.rebnconv3d(torch.cat((hx4dup,hx3),1))
+        hx3dup = _upsample_like(hx3d,hx2)
+        hx2d = self.rebnconv2d(torch.cat((hx3dup,hx2),1))
+        hx2dup = _upsample_like(hx2d,hx1)
+        hx1d = self.rebnconv1d(torch.cat((hx2dup,hx1),1))
+        return hx1d + hxin
+### RSU-4 ###
+class RSU4(nn.Module):
+    def __init__(self, in_ch=3, mid_ch=12, out_ch=3):
+        super(RSU4,self).__init__()
+        self.rebnconvin = REBNCONV(in_ch,out_ch,dirate=1)
+        self.rebnconv1 = REBNCONV(out_ch,mid_ch,dirate=1)
+        self.pool1 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv2 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.pool2 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.rebnconv3 = REBNCONV(mid_ch,mid_ch,dirate=1)
+        self.rebnconv4 = REBNCONV(mid_ch,mid_ch,dirate=2)
+        self.rebnconv3d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv2d = REBNCONV(mid_ch*2,mid_ch,dirate=1)
+        self.rebnconv1d = REBNCONV(mid_ch*2,out_ch,dirate=1)
+    def forward(self,x):
+        hx = x
+        hxin = self.rebnconvin(hx)
+        hx1 = self.rebnconv1(hxin)
+        hx = self.pool1(hx1)
+        hx2 = self.rebnconv2(hx)
+        hx = self.pool2(hx2)
+        hx3 = self.rebnconv3(hx)
+        hx4 = self.rebnconv4(hx3)
+        hx3d = self.rebnconv3d(torch.cat((hx4,hx3),1))
+        hx3dup = _upsample_like(hx3d,hx2)
+        hx2d = self.rebnconv2d(torch.cat((hx3dup,hx2),1))
+        hx2dup = _upsample_like(hx2d,hx1)
+        hx1d = self.rebnconv1d(torch.cat((hx2dup,hx1),1))
+        return hx1d + hxin
+### RSU-4F ###
+class RSU4F(nn.Module):
+    def __init__(self, in_ch=3, mid_ch=12, out_ch=3):
+        super(RSU4F,self).__init__()
+        self.rebnconvin = REBNCONV(in_ch,out_ch,dirate=1)
+        self.rebnconv1 = REBNCONV(out_ch,mid_ch,dirate=1)
+        self.rebnconv2 = REBNCONV(mid_ch,mid_ch,dirate=2)
+        self.rebnconv3 = REBNCONV(mid_ch,mid_ch,dirate=4)
+        self.rebnconv4 = REBNCONV(mid_ch,mid_ch,dirate=8)
+        self.rebnconv3d = REBNCONV(mid_ch*2,mid_ch,dirate=4)
+        self.rebnconv2d = REBNCONV(mid_ch*2,mid_ch,dirate=2)
+        self.rebnconv1d = REBNCONV(mid_ch*2,out_ch,dirate=1)
+    def forward(self,x):
+        hx = x
+        hxin = self.rebnconvin(hx)
+        hx1 = self.rebnconv1(hxin)
+        hx2 = self.rebnconv2(hx1)
+        hx3 = self.rebnconv3(hx2)
+        hx4 = self.rebnconv4(hx3)
+        hx3d = self.rebnconv3d(torch.cat((hx4,hx3),1))
+        hx2d = self.rebnconv2d(torch.cat((hx3d,hx2),1))
+        hx1d = self.rebnconv1d(torch.cat((hx2d,hx1),1))
+        return hx1d + hxin
+class myrebnconv(nn.Module):
+    def __init__(self, in_ch=3,
+                       out_ch=1,
+                       kernel_size=3,
+                       stride=1,
+                       padding=1,
+                       dilation=1,
+                       groups=1):
+        super(myrebnconv,self).__init__()
+        self.conv = nn.Conv2d(in_ch,
+                              out_ch,
+                              kernel_size=kernel_size,
+                              stride=stride,
+                              padding=padding,
+                              dilation=dilation,
+                              groups=groups)
+        self.bn = nn.BatchNorm2d(out_ch)
+        self.rl = nn.ReLU(inplace=True)
+    def forward(self,x):
+        return self.rl(self.bn(self.conv(x)))
+class BriaRMBG(nn.Module):
+    def __init__(self,in_ch=3,out_ch=1):
+        super(BriaRMBG,self).__init__()
+        self.conv_in = nn.Conv2d(in_ch,64,3,stride=2,padding=1)
+        self.pool_in = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.stage1 = RSU7(64,32,64)
+        self.pool12 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.stage2 = RSU6(64,32,128)
+        self.pool23 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.stage3 = RSU5(128,64,256)
+        self.pool34 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.stage4 = RSU4(256,128,512)
+        self.pool45 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.stage5 = RSU4F(512,256,512)
+        self.pool56 = nn.MaxPool2d(2,stride=2,ceil_mode=True)
+        self.stage6 = RSU4F(512,256,512)
+        # decoder
+        self.stage5d = RSU4F(1024,256,512)
+        self.stage4d = RSU4(1024,128,256)
+        self.stage3d = RSU5(512,64,128)
+        self.stage2d = RSU6(256,32,64)
+        self.stage1d = RSU7(128,16,64)
+        self.side1 = nn.Conv2d(64,out_ch,3,padding=1)
+        self.side2 = nn.Conv2d(64,out_ch,3,padding=1)
+        self.side3 = nn.Conv2d(128,out_ch,3,padding=1)
+        self.side4 = nn.Conv2d(256,out_ch,3,padding=1)
+        self.side5 = nn.Conv2d(512,out_ch,3,padding=1)
+        self.side6 = nn.Conv2d(512,out_ch,3,padding=1)
+        # self.outconv = nn.Conv2d(6*out_ch,out_ch,1)
+    def forward(self,x):
+        hx = x
+        hxin = self.conv_in(hx)
+        #hx = self.pool_in(hxin)
+        #stage 1
+        hx1 = self.stage1(hxin)
+        hx = self.pool12(hx1)
+        #stage 2
+        hx2 = self.stage2(hx)
+        hx = self.pool23(hx2)
+        #stage 3
+        hx3 = self.stage3(hx)
+        hx = self.pool34(hx3)
+        #stage 4
+        hx4 = self.stage4(hx)
+        hx = self.pool45(hx4)
+        #stage 5
+        hx5 = self.stage5(hx)
+        hx = self.pool56(hx5)
+        #stage 6
+        hx6 = self.stage6(hx)
+        hx6up = _upsample_like(hx6,hx5)
+        #-------------------- decoder --------------------
+        hx5d = self.stage5d(torch.cat((hx6up,hx5),1))
+        hx5dup = _upsample_like(hx5d,hx4)
+        hx4d = self.stage4d(torch.cat((hx5dup,hx4),1))
+        hx4dup = _upsample_like(hx4d,hx3)
+        hx3d = self.stage3d(torch.cat((hx4dup,hx3),1))
+        hx3dup = _upsample_like(hx3d,hx2)
+        hx2d = self.stage2d(torch.cat((hx3dup,hx2),1))
+        hx2dup = _upsample_like(hx2d,hx1)
+        hx1d = self.stage1d(torch.cat((hx2dup,hx1),1))
+        #side output
+        d1 = self.side1(hx1d)
+        d1 = _upsample_like(d1,x)
+        d2 = self.side2(hx2d)
+        d2 = _upsample_like(d2,x)
+        d3 = self.side3(hx3d)
+        d3 = _upsample_like(d3,x)
+        d4 = self.side4(hx4d)
+        d4 = _upsample_like(d4,x)
+        d5 = self.side5(hx5d)
+        d5 = _upsample_like(d5,x)
+        d6 = self.side6(hx6)
+        d6 = _upsample_like(d6,x)
+        return [F.sigmoid(d1), F.sigmoid(d2), F.sigmoid(d3), F.sigmoid(d4), F.sigmoid(d5), F.sigmoid(d6)],[hx1d,hx2d,hx3d,hx4d,hx5d,hx6]

foo.py ADDED Viewed

	@@ -0,0 +1,2 @@


1	+ def hello():
2	+ print("hello world")

input.jpg ADDED Viewed

requirements.txt CHANGED Viewed

@@ -1,26 +1,9 @@
-torch==2.4.0
-torchaudio==2.4.0
-torchvision==0.19.0
-transformers==4.31.0
-accelerate==0.21.0
-datasets==2.1.0
-einops==0.8.0
-huggingface_hub==0.19.4
-importlib_metadata==6.3.0
-librosa==0.9.2
-matplotlib==3.9.0
-numpy==1.23.0
-omegaconf==2.3.0
-packaging==24.1
-progressbar33==2.4
-protobuf==3.20.*
-safetensors==0.4.4
-sentencepiece==0.1.99
-scipy==1.8.0
-soundfile==0.12.1
-torchlibrosa==0.1.0
-tqdm==4.63.1
-wandb==0.12.14
-ipython==8.12.0
-gradio==4.3.0
-wavio==0.0.7

+gradio==4.16.0
+gradio_imageslider
+#torch
+#torchvision
+pillow
+numpy
+typing
+gitpython
+huggingface_hub