Orpheus-Pitches-Inpainter

Sleeping

File size: 17,920 Bytes

#==========================================================================
# https://huggingface.co/spaces/projectlosangeles/Orpheus-Pitches-Inpainter
#==========================================================================

print('=' * 70)
print('Orpheus Pitches Inpainter Gradio App')

print('=' * 70)
print('Loading core Orpheus Pitches Inpainter modules...')

import os
import copy

import time as reqtime
import datetime
from pytz import timezone

print('=' * 70)
print('Loading main Orpheus Pitches Inpainter modules...')

os.environ['USE_FLASH_ATTENTION'] = '1'

import torch

torch.set_float32_matmul_precision('high')
torch.backends.cuda.matmul.allow_tf32 = True # allow tf32 on matmul
torch.backends.cudnn.allow_tf32 = True # allow tf32 on cudnn
torch.backends.cuda.enable_flash_sdp(True)

from huggingface_hub import hf_hub_download

import TMIDIX

from midi_to_colab_audio import midi_to_colab_audio

from x_transformer_2_3_1 import *

import random

import tqdm

print('=' * 70)
print('Loading aux Orpheus Pitches Inpainter modules...')

import matplotlib.pyplot as plt

import gradio as gr
import spaces

print('=' * 70)
print('PyTorch version:', torch.__version__)
print('=' * 70)
print('Done!')
print('Enjoy! :)')
print('=' * 70)

#==================================================================================

MODEL_CHECKPOINT = 'Orpheus_Music_Transformer_Trained_Model_128497_steps_0.6934_loss_0.7927_acc.pth'
SOUNDFONT_PATH = 'SGM-v2.01-YamahaGrand-Guit-Bass-v2.7.sf2'

#==================================================================================

print('=' * 70)
print('Instantiating model...')

device_type = 'cuda'
dtype = 'bfloat16'

ptdtype = {'bfloat16': torch.bfloat16, 'float16': torch.float16}[dtype]
ctx = torch.amp.autocast(device_type=device_type, dtype=ptdtype)

SEQ_LEN = 8192
PAD_IDX = 18819

model = TransformerWrapper(num_tokens=PAD_IDX + 1,
                           max_seq_len=SEQ_LEN,
                           attn_layers=Decoder(
                               dim=2048,
                               depth=8,
                               heads=32,
                               rotary_pos_emb=True,
                               attn_flash=True
                          )
)

model = AutoregressiveWrapper(model, ignore_index=PAD_IDX, pad_value=PAD_IDX)

print('=' * 70)
print('Loading model checkpoint...')      

model_checkpoint = hf_hub_download(repo_id='asigalov61/Orpheus-Music-Transformer', filename=MODEL_CHECKPOINT)

model.load_state_dict(torch.load(model_checkpoint, map_location=device_type, weights_only=True))

model = torch.compile(model, mode='max-autotune')

model.to(device_type)
model.eval()

print('=' * 70)
print('Done!')
print('=' * 70)
print('Model will use', dtype, 'precision...')
print('=' * 70)

#==================================================================================

def load_midi(input_midi):

    raw_score = TMIDIX.midi2single_track_ms_score(input_midi)

    escore_notes = TMIDIX.advanced_score_processor(raw_score, return_enhanced_score_notes=True, apply_sustain=True)

    if escore_notes:
    
        escore_notes = TMIDIX.augment_enhanced_score_notes(escore_notes[0], sort_drums_last=True)
        
        dscore = TMIDIX.delta_score_notes(escore_notes)
        
        dcscore = TMIDIX.chordify_score([d[1:] for d in dscore])
        
        melody_chords = [18816]
        
        #=======================================================
        # MAIN PROCESSING CYCLE
        #=======================================================
        
        for i, c in enumerate(dcscore):
        
            delta_time = c[0][0]
        
            melody_chords.append(delta_time)
        
            for e in c:
            
                #=======================================================
                
                # Durations
                dur = max(1, min(255, e[1]))
        
                # Patches
                pat = max(0, min(128, e[5]))
                
                # Pitches
                ptc = max(1, min(127, e[3]))
                
                # Velocities
                # Calculating octo-velocity
                
                vel = max(8, min(127, e[4]))
                velocity = round(vel / 15)-1
                
                #=======================================================
                # FINAL NOTE SEQ
                #=======================================================
                
                # Writing final note
                pat_ptc = (128 * pat) + ptc 
                dur_vel = (8 * dur) + velocity
        
                melody_chords.extend([pat_ptc+256, dur_vel+16768]) # 18816
    
        
        print('Done!')
        print('=' * 70)
        print('Score hss', len(melody_chords), 'tokens')
        print('=' * 70)

        return melody_chords

    else:
        return None

#==================================================================================

@spaces.GPU
def Inpaint_Pitches(input_midi,
                    patches_to_inpaint,
                    inpaint_every_nth_note,
                    max_inpainted_pitch_dev,
                    max_inpaint_tries_per_note,
                    num_prime_tokens,
                    num_mem_tokens,
                    model_temperature,
                    model_sampling_top_k
                   ):

    #===============================================================================

    print('=' * 70)
    print('Req start time: {:%Y-%m-%d %H:%M:%S}'.format(datetime.datetime.now(PDT)))
    start_time = reqtime.time()
    print('=' * 70)

    if input_midi is not None:

        print('=' * 70)
        print('Requested settings:')
        print('=' * 70)
        fn = os.path.basename(input_midi)
        fn1 = fn.split('.')[0]
        print('Input MIDI file name:', fn)
        print('-' * 70)
    
        print('Patches to inpaint:', patches_to_inpaint)
        print('Inpaint every nth note:', inpaint_every_nth_note)
        print('Max inpainted pitch dev:', max_inpainted_pitch_dev)
        print('Max inpaint tries per note:', max_inpaint_tries_per_note)
        print('-' * 70)
        print('Number of prime tokens:', num_prime_tokens)
        print('Number of memory tokens:', num_mem_tokens)
        print('-' * 70)
        print('Model temperature:', model_temperature)
        print('Model top p:', model_sampling_top_k)
       
        print('=' * 70)
    
        #==================================================================

    

        print('Loading MIDI...')
    
        melody_chords = load_midi(input_midi.name)
    
        if melody_chords is not None:
            
            print('Sample score tokens', melody_chords[:10])
        
            #==================================================================
            
            print('=' * 70)
            print('Inpainting...')
        
            ipatches = [patch2number[instr] for instr in patches_to_inpaint]
            
            notes_counter = 0
            
            inpainted_song = melody_chords[:num_prime_tokens]
            
            for i, t in enumerate(melody_chords[num_prime_tokens:]):
            
                if 256 <= t < 16768:
            
                    old_patch = (t-256) // 128
                    old_pitch = (t-256) % 128
            
                    if old_patch in ipatches and notes_counter % inpaint_every_nth_note == 0:
                    
                        x = torch.LongTensor(inpainted_song[-num_mem_tokens:]).cuda()

                        tries = 0
                        new_pitch = -1
            
                        while (new_pitch > old_pitch + max_inpainted_pitch_dev or new_pitch < old_pitch - max_inpainted_pitch_dev) and tries < max_inpaint_tries_per_note:
                        
                            with ctx:
                                out = model.generate(x,
                                                     1,
                                                     temperature=model_temperature,
                                                     filter_logits_fn=top_k,
                                                     filter_kwargs={'k': model_sampling_top_k},
                                                     return_prime=False,
                                                     verbose=False)
                            
                            y = out.tolist()[0]
                
                            new_pitch = (y-256) % 128
            
                            tries += 1
            
                        if tries == max_inpaint_tries_per_note:
                            new_pitch = old_pitch
                            
                        new_patch_pitch_tok = (128 * old_patch) + new_pitch + 256
                
                        inpainted_song.append(new_patch_pitch_tok)
                        
                    else:
                        inpainted_song.append(t)
                        
                else:
                    inpainted_song.append(t)
            
                notes_counter += 1
        
            #==================================================================
           
            print('=' * 70)
            print('Done!')
            print('=' * 70)
            
            #===============================================================================
            
            print('Rendering results...')
            
            print('=' * 70)
            print('Sample INTs', inpainted_song[:15])
            print('=' * 70)
        
            song_f = []
            
            if len(inpainted_song) != 0:
            
                time = 0
                dur = 1
                vel = 90
                pitch = 60
                channel = 0
                patch = 0
            
                patches = [-1] * 16
            
                channels = [0] * 16
                channels[9] = 1
            
                for ss in inpainted_song:
            
                    if 0 <= ss < 256:
            
                        time += ss * 16
            
                    if 256 <= ss < 16768:
            
                        patch = (ss-256) // 128
            
                        if patch < 128:
            
                            if patch not in patches:
                              if 0 in channels:
                                  cha = channels.index(0)
                                  channels[cha] = 1
                              else:
                                  cha = 15
            
                              patches[cha] = patch
                              channel = patches.index(patch)
                            else:
                              channel = patches.index(patch)
            
                        if patch == 128:
                            channel = 9
            
                        pitch = (ss-256) % 128
            
            
                    if 16768 <= ss < 18816:
            
                        dur = ((ss-16768) // 8) * 16
                        vel = (((ss-16768) % 8)+1) * 15
            
                        song_f.append(['note', time, dur, channel, pitch, vel, patch])
            
                patches = [0 if x==-1 else x for x in patches]

            output_score, patches, overflow_patches = TMIDIX.patch_enhanced_score_notes(song_f)
        
            fn1 = "Orpheus-Pitches-Inpainter-Composition"
            
            detailed_stats = TMIDIX.Tegridy_ms_SONG_to_MIDI_Converter(output_score,
                                                                      output_signature = 'Orpheus Pitches Inpainter',
                                                                      output_file_name = fn1,
                                                                      track_name='Project Los Angeles',
                                                                      list_of_MIDI_patches=patches
                                                                      )
            
            new_fn = fn1+'.mid'
                    
            
            audio = midi_to_colab_audio(new_fn, 
                                soundfont_path=SOUNDFONT_PATH,
                                sample_rate=16000,
                                volume_scale=10,
                                output_for_gradio=True
                                )
            
            print('Done!')
            print('=' * 70)
        
            #========================================================
        
            output_midi = str(new_fn)
            output_audio = (16000, audio)
            output_plot = TMIDIX.plot_ms_SONG(song_f, plot_title=output_midi, return_plt=True)
        
            print('Output MIDI file name:', output_midi)
            print('=' * 70) 
            
            #========================================================
    
        else:
            return None, None, None
    
        print('-' * 70)
        print('Req end time: {:%Y-%m-%d %H:%M:%S}'.format(datetime.datetime.now(PDT)))
        print('-' * 70)
        print('Req execution time:', (reqtime.time() - start_time), 'sec')
    
        return output_audio, output_plot, output_midi

    else:
        return None, None, None
    
#==================================================================================

PDT = timezone('US/Pacific')

print('=' * 70)
print('App start time: {:%Y-%m-%d %H:%M:%S}'.format(datetime.datetime.now(PDT)))
print('=' * 70)

#==================================================================================

patch2number = {v: k for k, v in TMIDIX.Number2patch.items()}

#==================================================================================

with gr.Blocks() as demo:

    #==================================================================================

    gr.Markdown("<h1 style='text-align: left; margin-bottom: 1rem'>Orpheus Pitches Inpainter</h1>")
    gr.Markdown("<h1 style='text-align: left; margin-bottom: 1rem'>Inpaint pitches in any MIDI composition</h1>")
    gr.HTML("""            
            <p> 
                <a href="https://huggingface.co/spaces/projectlosangeles/Orpheus-Pitches-Inpainter?duplicate=true">
                    <img src="https://huggingface.co/datasets/huggingface/badges/resolve/main/duplicate-this-space-md.svg" alt="Duplicate in Hugging Face">
                </a>
            </p>
            
            for faster execution and endless generation!
            """)
    
    #==================================================================================
    
    gr.Markdown("## Upload source MIDI or select a sample MIDI on the bottom of the page")
    
    input_midi = gr.File(label="Input MIDI", 
                         file_types=[".midi", ".mid", ".kar"]
                        )
    
    gr.Markdown("## Generation options")

    patches_to_inpaint = gr.Dropdown(label="Select instruments to inpaint", choices=list(patch2number.keys()),
                                     multiselect=True, type="value",
                                     info="Instruments MUST be present in the composition. For best results select a single instrument."
                                    )

    inpaint_every_nth_note = gr.Slider(1, 10, value=1, step=1, label="Inpaint every nth note")
    max_inpainted_pitch_dev = gr.Slider(12, 24, value=12, step=12, label="Maximum inpainted pitch deviation")
    max_inpaint_tries_per_note = gr.Slider(5, 100, value=10, step=1, label="Maximum inpainting attempts per note")

    num_prime_tokens = gr.Slider(0, 512, value=128, step=1, label="Number of prime tokens")
    num_mem_tokens = gr.Slider(32, 8192, value=4096, step=8, label="Number of prime tokens")

    model_temperature = gr.Slider(0.1, 1, value=0.9, step=0.01, label="Model temperature")
    model_sampling_top_k = gr.Slider(1, 100, value=15, step=1, label="Model sampling top k value")
    
    generate_btn = gr.Button("Generate", variant="primary")

    gr.Markdown("## Generation results")

    output_title = gr.Textbox(label="MIDI melody title")
    output_audio = gr.Audio(label="MIDI audio", format="wav", elem_id="midi_audio")
    output_plot = gr.Plot(label="MIDI score plot")
    output_midi = gr.File(label="MIDI file", file_types=[".mid"])

    generate_btn.click(Inpaint_Pitches, 
                       [input_midi,
                        patches_to_inpaint,
                        inpaint_every_nth_note,
                        max_inpainted_pitch_dev,
                        max_inpaint_tries_per_note,
                        num_prime_tokens,
                        num_mem_tokens,
                        model_temperature,
                        model_sampling_top_k
                       ],
                       [output_audio,
                        output_plot,
                        output_midi                          
                       ]
                      )

    gr.Examples(
                [["Orpheus-Music-Transformer-MI-Seed-1.mid", "Clarinet", 1, 12, 10, 128, 4096, 0.9, 15]
                ],
                [input_midi,
                 patches_to_inpaint,
                 inpaint_every_nth_note,
                 max_inpainted_pitch_dev,
                 max_inpaint_tries_per_note,
                 num_prime_tokens,
                 num_mem_tokens,
                 model_temperature,
                 model_sampling_top_k
                ],
                [output_audio,
                 output_plot,
                 output_midi
                ],
                Inpaint_Pitches
    )
    
#==================================================================================

demo.launch()

#==================================================================================