Team Ai
Apppublic

avans06/Audio_Spectrogram_Video_Generator

sourceHugging Faceupdated 1y agoView on Hugging Face
1likes
app.py935 linesDownload Raw Back to root
1import gradio as gr2import librosa3import numpy as np4import re5import os6import time7import struct8import subprocess9import soundfile as sf10import matplotlib.font_manager as fm11from PIL import ImageFont12from typing import Tuple, List, Dict, Set13from mutagen.flac import FLAC14from moviepy import CompositeVideoClip, TextClip, VideoClip, AudioFileClip, ImageClip15 16# --- Font Scanning and Management ---17def get_font_display_name(font_path: str) -> Tuple[str, str]:18    """19    A robust TTF/TTC parser based on the user's final design.20    It reads the 'name' table to find the localized "Full Font Name" (nameID=4).21    Returns a tuple of (display_name, language_tag {'zh'/'ja'/'ko'/'en'/'other'}).22    """23    def decode_name_string(name_bytes: bytes, platform_id: int, encoding_id: int) -> str:24        """Decodes the name string based on platform and encoding IDs."""25        try:26            if platform_id == 3 and encoding_id in [1, 10]: # Windows, Unicode27                return name_bytes.decode('utf_16_be').strip('\x00')28            elif platform_id == 1 and encoding_id == 0: # Macintosh, Roman29                return name_bytes.decode('mac_roman').strip('\x00')30            elif platform_id == 0: # Unicode31                return name_bytes.decode('utf_16_be').strip('\x00')32            else: # Fallback33                return name_bytes.decode('utf_8', errors='ignore').strip('\x00')34        except Exception:35            return None36 37    try:38        with open(font_path, 'rb') as f: data = f.read()39        def read_ushort(offset):40            return struct.unpack('>H', data[offset:offset+2])[0]41        def read_ulong(offset):42            return struct.unpack('>I', data[offset:offset+4])[0]43        font_offsets = [0]44        # Check for TTC (TrueType Collection) header45        if data[:4] == b'ttcf':46            num_fonts = read_ulong(8)47            font_offsets = [read_ulong(12 + i * 4) for i in range(num_fonts)]48        49        # For simplicity, we only parse the first font in a TTC50        font_offset = font_offsets[0]51 52        num_tables = read_ushort(font_offset + 4)53        name_table_offset = -154        # Locate the 'name' table55        for i in range(num_tables):56            entry_offset = font_offset + 12 + i * 1657            tag = data[entry_offset:entry_offset+4]58            if tag == b'name':59                name_table_offset = read_ulong(entry_offset + 8)60                break61 62        if name_table_offset == -1:63            return None, None64 65        count, string_offset = read_ushort(name_table_offset + 2), read_ushort(name_table_offset + 4)66        name_candidates = {}67        # Iterate through all name records68        for i in range(count):69            rec_offset = name_table_offset + 6 + i * 1270            platform_id, encoding_id, language_id, name_id, length, offset = struct.unpack('>HHHHHH', data[rec_offset:rec_offset+12])71            72            if name_id == 4:  # We only care about the "Full Font Name"73                string_pos = name_table_offset + string_offset + offset74                value = decode_name_string(data[string_pos : string_pos + length], platform_id, encoding_id)75 76                if value:77                    # Store candidates based on language ID78                    if language_id in [1028, 2052, 3076, 4100, 5124]:79                        name_candidates["zh"] = value80                    elif language_id == 1041:81                        name_candidates["ja"] = value82                    elif language_id == 1042:83                        name_candidates["ko"] = value84                    elif language_id in [1033, 0]:85                        name_candidates["en"] = value86                    else:87                        if "other" not in name_candidates:88                            name_candidates["other"] = value89                            90        # Return the best candidate based on language priority91        if name_candidates.get("zh"):92            return name_candidates.get("zh"), "zh"93        if name_candidates.get("ja"):94            return name_candidates.get("ja"), "ja"95        if name_candidates.get("ko"):96            return name_candidates.get("ko"), "ko"97        if name_candidates.get("other"):98            return name_candidates.get("other"), "other"99        if name_candidates.get("en"):100            return name_candidates.get("en"), "en"101        return None, None102 103    except Exception:104        return None, None105 106def get_font_data() -> Tuple[Dict[str, str], List[str]]:107    """108    Scans system fonts, parses their display names, and returns a sorted list109    with a corresponding name-to-path map.110    """111    font_map = {}112    found_names = [] # Stores (display_name, is_fallback, lang_tag)113    114    # Scan for both .ttf and .ttc files115    ttf_files = fm.findSystemFonts(fontpaths=None, fontext='ttf')116    ttc_files = fm.findSystemFonts(fontpaths=None, fontext='ttc')117    all_font_files = list(set(ttf_files + ttc_files))118    119    for path in all_font_files:120        display_name, lang_tag = get_font_display_name(path)121        is_fallback = display_name is None122 123        if is_fallback:124            # Create a fallback name from the filename125            display_name = os.path.splitext(os.path.basename(path))[0].replace('-', ' ').replace('_', ' ').title()126            lang_tag = 'fallback'127 128        if display_name and display_name not in font_map:129            font_map[display_name] = path130            found_names.append((display_name, is_fallback, lang_tag))131 132    # Define sort priority for languages133    sort_order = {'zh': 0, 'ja': 1, 'ko': 2, 'en': 3, 'other': 4, 'fallback': 5}134    135    # Sort by priority, then alphabetically136    found_names.sort(key=lambda x: (sort_order.get(x[2], 99), x[0]))137 138    sorted_display_names = [name for name, _, _ in found_names]139    return font_map, sorted_display_names140 141print("Scanning system fonts and parsing names...")142SYSTEM_FONTS_MAP, FONT_DISPLAY_NAMES = get_font_data()143print(f"Scan complete. Found {len(FONT_DISPLAY_NAMES)} available fonts.")144 145 146# --- CUE Sheet Parsing Logic ---147def cue_time_to_seconds(time_str: str) -> float:148    try:149        minutes, seconds, frames = map(int, time_str.split(':'))150        return minutes * 60 + seconds + frames / 75.0151    except ValueError:152        return 0.0153 154def parse_cue_sheet_manually(cue_data: str) -> List[Dict[str, any]]:155    tracks = []156    current_track_info = None157    for line in cue_data.splitlines():158        line = line.strip()159        if line.upper().startswith('TRACK'):160            if current_track_info and 'title' in current_track_info and 'start_time' in current_track_info:161                tracks.append(current_track_info)162            current_track_info = {}163            continue164        if current_track_info is not None:165            title_match = re.search(r'TITLE\s+"(.*?)"', line, re.IGNORECASE)166            if title_match:167                current_track_info['title'] = title_match.group(1)168                continue169            index_match = re.search(r'INDEX\s+01\s+(\d{2}:\d{2}:\d{2})', line, re.IGNORECASE)170            if index_match:171                current_track_info['start_time'] = cue_time_to_seconds(index_match.group(1))172                continue173    if current_track_info and 'title' in current_track_info and 'start_time' in current_track_info:174        tracks.append(current_track_info)175    return tracks176 177 178# --- FFmpeg Framerate Conversion ---179def increase_video_framerate(input_path: str, output_path: str, target_fps: int = 24):180    """181    Uses FFmpeg to increase the video's framerate without re-encoding.182    This is extremely fast as it only copies streams and changes metadata.183    184    Args:185        input_path (str): Path to the low-framerate video file.186        output_path (str): Path for the final, high-framerate video file.187        target_fps (int): The desired output framerate.188    """189    print(f"Increasing framerate of '{input_path}' to {target_fps} FPS...")190    191    # Construct the FFmpeg command based on the user's specification192    command = [193        'ffmpeg',194        '-y',  # Overwrite output file if exists195        '-i', input_path,196        '-map', '0',                # Map all streams (video, audio, subtitles)197        '-vf', f'fps={target_fps}', # Use fps filter to convert framerate to 24198        '-c:v', 'libx264',          # Re-encode video with H.264 codec199        '-preset', 'fast',          # Encoding speed/quality tradeoff200        '-crf', '18',               # Quality (lower is better)201        '-c:a', 'copy',             # Copy audio without re-encoding202        output_path203    ]204 205    try:206        # Execute the command207        # Using capture_output to hide ffmpeg logs from the main console unless an error occurs208        result = subprocess.run(command, check=True, capture_output=True, text=True)209        print("Framerate increase successful.")210    except FileNotFoundError:211        # This error occurs if FFmpeg is not installed or not in the system's PATH212        raise gr.Error("FFmpeg not found. Please ensure FFmpeg is installed and accessible in your system's PATH.")213    except subprocess.CalledProcessError as e:214        # This error occurs if FFmpeg returns a non-zero exit code215        print("FFmpeg error output:\n", e.stderr)216        raise gr.Error(f"FFmpeg failed to increase the framerate. See console for details. Error: {e.stderr}")217 218 219# --- HELPER FUNCTION for parsing track ranges ---220def parse_track_ranges(range_str: str) -> Set[int]:221    """Parses a string like '1-4, 7, 10-13' into a set of integers."""222    if not range_str:223        return set()224    225    indices = set()226    parts = range_str.split(',')227    for part in parts:228        part = part.strip()229        if not part:230            continue231        if '-' in part:232            try:233                start, end = map(int, part.split('-'))234                indices.update(range(start, end + 1))235            except ValueError:236                print(f"Warning: Could not parse range '{part}'. Skipping.")237        else:238            try:239                indices.add(int(part))240            except ValueError:241                print(f"Warning: Could not parse track number '{part}'. Skipping.")242    return indices243 244 245# --- Main Processing Function ---246def process_audio_to_video(*args, progress=gr.Progress(track_tqdm=True)):247    # --- Correctly unpack all arguments from *args using slicing ---248    MAX_GROUPS = 10  # This MUST match the UI definition249 250    # Define the structure of the *args tuple based on the `all_inputs` list251    audio_files = args[0]252    253    # Slice the args tuple to get the continuous blocks of inputs254    all_track_strs = args[1 : 1 + MAX_GROUPS]255    all_image_lists = args[1 + MAX_GROUPS : 1 + MAX_GROUPS * 2]256    257    # Group inputs are packed in pairs (track_str, image_list)258    group_definitions = []259    for i in range(MAX_GROUPS):260        group_definitions.append({261            "tracks_str": all_track_strs[i],262            "images": all_image_lists[i]263        })264 265    # Unpack the remaining arguments with correct indexing266    arg_offset = 1 + MAX_GROUPS * 2267    fallback_images = args[arg_offset]268    format_double_digits = args[arg_offset + 1]269    video_width = args[arg_offset + 2]270    video_height = args[arg_offset + 3]271    spec_fg_color = args[arg_offset + 4]272    spec_bg_color = args[arg_offset + 5]273    274    # --- NEW: Unpack spectrogram style arguments ---275    n_bands = int(args[arg_offset + 6])276    bar_spacing = int(args[arg_offset + 7])277    mirror_mode = args[arg_offset + 8] # This is now a string278    bar_style = args[arg_offset + 9]279    num_blocks = int(args[arg_offset + 10])280    281    # --- Unpack font and text arguments (indices are shifted) ---282    font_name = args[arg_offset + 11]283    font_size = args[arg_offset + 12]284    font_color = args[arg_offset + 13]285    font_bg_color = args[arg_offset + 14]286    font_bg_alpha = args[arg_offset + 15]287    pos_h = args[arg_offset + 16]288    pos_v = args[arg_offset + 17]289 290 291    if not audio_files:292        raise gr.Error("Please upload at least one audio file.")293    if not font_name:294        raise gr.Error("Please select a font from the list.")295    296    progress(0, desc="Initializing...")297    298    # Define paths for temporary and final files299    timestamp = int(time.time())300    temp_fps1_path = f"temp_{timestamp}_fps1.mp4"301    temp_audio_path = f"temp_combined_audio_{timestamp}.wav"302    final_output_path = f"final_video_{timestamp}_fps24.mp4"303 304    WIDTH, HEIGHT = int(video_width), int(video_height)305    RENDER_FPS = 1 # Render at 1 FPS306    PLAYBACK_FPS = 24 # Final playback framerate307    308    # --- A robust color parser for hex and rgb() strings ---309    def parse_color_to_rgb(color_str: str) -> Tuple[int, int, int]:310        """311        Parses a color string which can be in hex format (#RRGGBB) or312        rgb format (e.g., "rgb(255, 128, 0)").313        Returns a tuple of (R, G, B).314        """315        color_str = color_str.strip()316        if color_str.startswith('#'):317            # Handle hex format318            hex_val = color_str.lstrip('#')319            if len(hex_val) == 3: # Handle shorthand hex like #FFF320                hex_val = "".join([c*2 for c in hex_val])321            return tuple(int(hex_val[i:i+2], 16) for i in (0, 2, 4))322        elif color_str.startswith('rgb'):323            # Handle rgb format324            try:325                numbers = re.findall(r'\d+', color_str)326                return tuple(int(n) for n in numbers[:3])327            except (ValueError, IndexError):328                raise ValueError(f"Could not parse rgb color string: {color_str}")329        else:330            raise ValueError(f"Unknown color format: {color_str}")331        332    # Use the new robust parser for all color inputs333    fg_rgb, bg_rgb = parse_color_to_rgb(spec_fg_color), parse_color_to_rgb(spec_bg_color)334    grid_rgb = tuple(min(c + 40, 255) for c in bg_rgb)335    336    # Wrap the entire process in a try...finally block to ensure cleanup337    try:338        # --- Define total steps for the progress bar ---339        TOTAL_STEPS = 5340        341        # --- Stage 1: Audio Processing & Master Track List Creation ---342        master_track_list, y_accumulator, current_sr = [], [], None343        total_duration, global_track_counter = 0.0, 0344        345        # --- Use `progress.tqdm` to create a progress bar for this loop ---346        for file_idx, audio_path in enumerate(progress.tqdm(audio_files, desc=f"Stage 1/{TOTAL_STEPS}: Analyzing Audio Files")):347            # --- Load audio as stereo (or its original channel count) ---348            y, sr = librosa.load(audio_path, sr=None, mono=False)349            # If loaded audio is mono (1D array), convert it to a 2D stereo array350            # by duplicating the channel. This ensures all arrays can be concatenated.351            if y.ndim == 1:352                print(f"  - Converting mono file to stereo: {os.path.basename(audio_path)}")353                y = np.stack([y, y])354            355            if current_sr is None:356                current_sr = sr357            if current_sr != sr:358                print(f"Warning: Sample rate mismatch for {os.path.basename(audio_path)}. Expected {current_sr}Hz, found {sr}Hz.")359                print(f"Resampling from {sr}Hz to {current_sr}Hz...")360                y = librosa.resample(y, orig_sr=sr, target_sr=current_sr)361            362            y_accumulator.append(y)363            # Use the first channel (y[0]) for duration calculation, which is standard practice364            file_duration = librosa.get_duration(y=y[0], sr=current_sr)365            366            # First, try to parse the CUE sheet from the audio file.367            cue_tracks = []368            if audio_path.lower().endswith('.flac'):369                try:370                    audio_meta = FLAC(audio_path)371                    if 'cuesheet' in audio_meta.tags:372                        cue_tracks = parse_cue_sheet_manually(audio_meta.tags['cuesheet'][0])373                        374                        print(f"Successfully parsed {len(cue_tracks)} tracks from CUE sheet.")375                except Exception as e:376                    print(f"Warning: Could not parse CUE sheet for {os.path.basename(audio_path)}: {e}")377 378            if cue_tracks:379                for track_idx, track in enumerate(cue_tracks):380                    global_track_counter += 1381                    start_time = track.get('start_time', 0)382                    end_time = cue_tracks[track_idx+1].get('start_time', file_duration) if track_idx + 1 < len(cue_tracks) else file_duration383                    master_track_list.append({"global_index": global_track_counter, "title": track.get('title', 'Unknown'), "start_time": total_duration + start_time, "end_time": total_duration + end_time})384            else:385                global_track_counter += 1386                master_track_list.append({"global_index": global_track_counter, "title": os.path.splitext(os.path.basename(audio_path))[0], "start_time": total_duration, "end_time": total_duration + file_duration})387            388            total_duration += file_duration389            390        # --- Concatenate along the time axis (axis=1) for stereo arrays ---391        y_combined = np.concatenate(y_accumulator, axis=1)392        duration = total_duration393        394        # --- Transpose the array for soundfile to write stereo correctly ---395        sf.write(temp_audio_path, y_combined.T, current_sr)396        print(f"Combined all audio files into one. Total duration: {duration:.2f}s")397        398        # --- Update progress to the next stage, use fractional progress (current/total) ---399        progress(1 / TOTAL_STEPS, desc=f"Stage 2/{TOTAL_STEPS}: Mapping Images to Tracks")400        401        # --- Stage 2: Map Tracks to Image Groups ---402        parsed_groups = [parse_track_ranges(g['tracks_str']) for g in group_definitions]403        track_to_images_map = {}404        for track_info in master_track_list:405            track_idx = track_info['global_index']406            assigned = False407            for i, group_indices in enumerate(parsed_groups):408                if track_idx in group_indices:409                    track_to_images_map[track_idx] = group_definitions[i]['images']410                    assigned = True411                    break412            if not assigned:413                track_to_images_map[track_idx] = fallback_images414        415        # --- Stage 3: Generate ImageClips based on contiguous blocks ---416        image_clips = []417        if any(track_to_images_map.values()):418            current_track_cursor = 0419            while current_track_cursor < len(master_track_list):420                start_track_info = master_track_list[current_track_cursor]421                image_set_for_block = track_to_images_map.get(start_track_info['global_index'])422 423                # Find the end of the contiguous block of tracks that use the same image set424                end_track_cursor = current_track_cursor425                while (end_track_cursor + 1 < len(master_track_list) and426                       track_to_images_map.get(master_track_list[end_track_cursor + 1]['global_index']) == image_set_for_block):427                    end_track_cursor += 1428 429                end_track_info = master_track_list[end_track_cursor]430                431                block_start_time = start_track_info['start_time']432                block_end_time = end_track_info['end_time']433                block_duration = block_end_time - block_start_time434 435                if image_set_for_block and block_duration > 0:436                    print(f"Creating image block for tracks {start_track_info['global_index']}-{end_track_info['global_index']} (Time: {block_start_time:.2f}s - {block_end_time:.2f}s)")437                    time_per_image = block_duration / len(image_set_for_block)438                    for i, img_path in enumerate(image_set_for_block):439                        def create_image_layer(path, start, dur):440                            try:441                                img = ImageClip(path)442                                scale = min(WIDTH/img.w, HEIGHT/img.h)443                                resized_img = img.resized(scale)444                                return CompositeVideoClip([resized_img.with_position("center")], size=(WIDTH, HEIGHT)).with_duration(dur).with_start(start)445                            except Exception as e:446                                print(f"Warning: Failed to process image '{path}'. Skipping. Error: {e}")447                                return None448 449                        clip = create_image_layer(img_path, block_start_time + i * time_per_image, time_per_image)450                        if clip:451                            image_clips.append(clip)452 453                current_track_cursor = end_track_cursor + 1454 455        progress(2 / TOTAL_STEPS, desc=f"Stage 3/{TOTAL_STEPS}: Generating Text & Spectrogram")456 457        # --- Stage 4: Generate Text and Spectrogram ---458        # --- Text Overlay Logic using the aggregated track info459        text_clips = [] # Text clips are now simpler as they don't depend on complex file logic anymore460        461        font_path = SYSTEM_FONTS_MAP.get(font_name)462        if not font_path:463            raise gr.Error(f"Font path for '{font_name}' not found!")464        465        # Use the robust parser for text colors as well466        font_bg_rgb = parse_color_to_rgb(font_bg_color)467 468        position = (pos_h.lower(), pos_v.lower())469        470        print(f"Using font: {font_name}, Size: {font_size}, Position: {position}")471 472        # Create the RGBA tuple for the background color.473        # The alpha value is converted from a 0.0-1.0 float to a 0-255 integer.474        bg_color_tuple = (font_bg_rgb[0], font_bg_rgb[1], font_bg_rgb[2], int(font_bg_alpha * 255))475        476        # 1. Define a maximum width for the caption. 90% of the video width is a good choice.477        caption_width = int(WIDTH * 0.9)478            479        # --- Get font metrics to calculate dynamic padding ---480        try:481            # Load the font with Pillow to access its metrics482            pil_font = ImageFont.truetype(font_path, size=font_size)483            _, descent = pil_font.getmetrics()484            # Calculate a bottom margin to compensate for the font's descent.485            # A small constant is added as a safety buffer.486            # This prevents clipping on fonts with large descenders (like 'g', 'p').487            bottom_margin = int(descent * 0.5) + 2488            print(f"Font '{font_name}' descent: {descent}. Applying dynamic bottom margin of {bottom_margin}px.")489        except Exception as e:490            # Fallback in case of any font loading error491            print(f"Warning: Could not get font metrics for '{font_name}'. Using fixed margin. Error: {e}")492            bottom_margin = int(WIDTH * 0.01) # A small fixed fallback493            494        for track in master_track_list:495            text_duration = track['end_time'] - track['start_time']496            if text_duration <= 0:497                continue498            499            # Construct display text based on pre-formatted number string500            num_str = f"{track['global_index']:02d}" if format_double_digits else str(track['global_index'])501            display_text = f"{num_str}. {track['title']}"502 503            504            # 1. Create the TextClip first without positioning to get its size505            txt_clip = TextClip(506                text=display_text.strip(),507                font_size=font_size,508                color=font_color,509                font=font_path,510                bg_color=bg_color_tuple,511                method='caption', # <-- Set method to caption512                size=(caption_width, None), # <-- Provide size for wrapping513                margin=(0, 0, 0, bottom_margin)514            ).with_position(position).with_duration(text_duration).with_start(track['start_time'])515 516            text_clips.append(txt_clip)517 518        N_FFT, HOP_LENGTH = 2048, 512519        MIN_DB, MAX_DB = -80.0, 0.0520 521        # Spectrogram calculation on combined audio522        # --- Create a mono version of audio specifically for the spectrogram ---523        # This resolves the TypeError while keeping the final audio in stereo.524        y_mono_for_spec = librosa.to_mono(y_combined)525        S_mel = librosa.feature.melspectrogram(y=y_mono_for_spec, sr=current_sr, n_fft=N_FFT, hop_length=HOP_LENGTH, n_mels=n_bands, fmax=current_sr/2)526        S_mel_db = librosa.power_to_db(S_mel, ref=np.max)527        528        # --- Pre-calculate drawing parameters for stacked block style ---529        BLOCK_SPACING = 2 # The pixel gap between stacked blocks530        if bar_style == 'Stacked Blocks':531            # Calculate the total vertical space available for the blocks themselves532            # In mirrored mode, this is based on half the screen height533            if mirror_mode == 'Vertical (Left/Right)':534                drawable_size = WIDTH // 2535            elif mirror_mode == 'Horizontal (Top/Bottom)':536                drawable_size = HEIGHT // 2537            else: # Off538                drawable_size = HEIGHT539            total_block_pixel_size = drawable_size - ((num_blocks - 1) * BLOCK_SPACING)540            # Calculate the size of a single block541            single_block_size = total_block_pixel_size / num_blocks542 543        # Frame generation logic for the spectrogram544        def frame_generator(t):545            # If images are used as background, the spectrogram's own background should be transparent.546            # Otherwise, use the selected background color.547            # Here, we will use a simple opacity setting on the final clip, so we always generate the frame.548            frame_bg = bg_rgb if not image_clips else (0,0,0) # Use black if it will be made transparent later549            frame = np.full((HEIGHT, WIDTH, 3), frame_bg, dtype=np.uint8)550 551            # Draw the grid lines only if no images are being used.552            if not image_clips:553                for i in range(1, 9):554                    y_pos = int(i * (HEIGHT / 9)); frame[y_pos-1:y_pos, :] = grid_rgb555 556            # 1. Safety Check: If the spectrogram has no time frames (e.g., from an extremely short audio file),557            #    return a blank frame immediately to prevent an IndexError.558            if S_mel_db.shape[1] == 0:559                return frame560 561            # 2. Use librosa.time_to_frames to accurately convert the video time `t`562            #    into a spectrogram frame index. This is far more reliable than manual scaling563            #    and solves the problem of missing content on the rightmost side of the video.564            time_idx = librosa.time_to_frames(t, sr=current_sr, hop_length=HOP_LENGTH)565            566            # 3. Boundary Protection: Although time_to_frames is accurate, this extra `min`567            #    call acts as a safeguard to ensure the index never exceeds the array's568            #    maximum valid index, preventing any edge-case errors.569            time_idx = min(time_idx, S_mel_db.shape[1] - 1)570            571            # --- RENDER LOGIC FOR VERTICAL MIRROR ---572            if mirror_mode == 'Vertical (Left/Right)':573                center_x = WIDTH // 2574                max_pixel_length = WIDTH // 2575                bar_height = HEIGHT / n_bands576 577                for i in range(n_bands):578                    energy_db = S_mel_db[i, time_idx]579                    norm_height = np.clip((energy_db - MIN_DB) / (MAX_DB - MIN_DB), 0, 1)580                    if norm_height == 0:581                        continue582 583                    # --- Calculate y-coords from bottom-to-top ---584                    # This makes low frequencies appear at the bottom and high frequencies at the top.585                    y_start = int(HEIGHT - (i + 1) * bar_height)586                    y_end = int(HEIGHT - i * bar_height)587                    588                    # Apply spacing to create a gap above the current bar.589                    y_start_with_spacing = y_start + bar_spacing590 591                    # Ensure the bar still has visible height after spacing592                    if y_start_with_spacing >= y_end:593                        continue594 595                    if bar_style == 'Stacked Blocks':596                        blocks_to_draw = int(norm_height * num_blocks)597                        if blocks_to_draw == 0:598                            continue599 600                        for j in range(blocks_to_draw):601                            block_left_x = center_x + (j * (single_block_size + BLOCK_SPACING))602                            block_right_x = block_left_x + single_block_size603                            # Draw right side604                            frame[y_start_with_spacing:y_end, int(block_left_x):int(block_right_x)] = fg_rgb605                            # Draw mirrored left side606                            frame[y_start_with_spacing:y_end, int(center_x - (block_right_x - center_x)):int(center_x - (block_left_x - center_x))] = fg_rgb607                    else: # Solid Bars608                        bar_pixel_length = int(norm_height * max_pixel_length)609                        if bar_pixel_length < 1:610                            continue611 612                        # Draw right side613                        frame[y_start_with_spacing:y_end, center_x : center_x + bar_pixel_length] = fg_rgb614                        # Draw mirrored left side615                        frame[y_start_with_spacing:y_end, center_x - bar_pixel_length : center_x] = fg_rgb616            617            # --- RENDER LOGIC FOR HORIZONTAL MIRROR AND OFF ---618            else:619                bar_width = WIDTH / n_bands620                is_horizontal_mirror = (mirror_mode == 'Horizontal (Top/Bottom)')621                622                # Determine rendering parameters based on whether the view is mirrored623                if is_horizontal_mirror:624                    center_y = HEIGHT // 2625                    max_pixel_height = HEIGHT // 2626                else: # Off627                    center_y = HEIGHT # The "center" is the bottom of the screen628                    max_pixel_height = HEIGHT629 630                # Loop through each frequency band to draw its bar/blocks631                for i in range(n_bands):632                    energy_db = S_mel_db[i, time_idx]633                634                    # The denominator should be the range of DB values (MAX_DB - MIN_DB).635                    # Since MAX_DB is 0, this simplifies to -MIN_DB, which is a positive 80.0.636                    # This prevents the division by zero warning.637                    norm_height = np.clip((energy_db - MIN_DB) / (MAX_DB - MIN_DB), 0, 1)638                    639                    if norm_height == 0:640                        continue641 642                    # Calculate the horizontal position of the current bar643                    x_start = int(i * bar_width)644                    x_end = int((i + 1) * bar_width - bar_spacing)645 646                    # --- Main rendering logic: switches between styles ---647                    if bar_style == 'Stacked Blocks':648                        # Calculate how many blocks to draw based on energy649                        blocks_to_draw = int(norm_height * num_blocks)650                        if blocks_to_draw == 0:651                            continue652                    653                        # Draw each block from the bottom up654                        for j in range(blocks_to_draw):655                            # Calculate the Y coordinates for this specific block656                            block_bottom_y = center_y - (j * (single_block_size + BLOCK_SPACING))657                            block_top_y = block_bottom_y - single_block_size658                            frame[int(block_top_y):int(block_bottom_y), x_start:x_end] = fg_rgb659                            660                            if is_horizontal_mirror:661                                frame[int(center_y + (center_y - block_bottom_y)):int(center_y + (center_y - block_top_y)), x_start:x_end] = fg_rgb662                    else: # Solid Bars663                        # Calculate the total height of the solid bar664                        bar_pixel_height = int(norm_height * max_pixel_height)665                        666                        if bar_pixel_height < 1:667                            continue668 669                        frame[center_y - bar_pixel_height : center_y, x_start:x_end] = fg_rgb670                        671                        if is_horizontal_mirror:672                            frame[center_y : center_y + bar_pixel_height, x_start:x_end] = fg_rgb673            return frame674            675        video_clip = VideoClip(frame_function=frame_generator, duration=duration)676        677        # --- Set Spectrogram Opacity ---678        # If image clips were created, make the spectrogram layer 50% transparent.679        if image_clips:680            print("Applying 50% opacity to spectrogram layer.")681            video_clip = video_clip.with_opacity(0.5)682            683        # --- Use fractional progress (current/total) ---684        progress(3 / TOTAL_STEPS, desc=f"Stage 4/{TOTAL_STEPS}: Rendering Base Video")685        686        # --- Composition and Rendering ---687        audio_clip = AudioFileClip(temp_audio_path)688        689        # --- Clip Composition ---690        # The final composition order is important: images at the bottom, then spectrogram, then text.691        # The base layer is now the list of image clips.692        final_layers = image_clips + [video_clip] + text_clips693        final_clip = CompositeVideoClip(final_layers, size=(WIDTH, HEIGHT)).with_audio(audio_clip)694        695        # Step 1: Render the slow, 1 FPS intermediate file696        print(f"Step 1/2: Rendering base video at {RENDER_FPS} FPS...")697        try:698            # Attempt to copy audio stream directly699            print("Attempting to copy audio stream directly...")700            final_clip.write_videofile(701                temp_fps1_path, codec="libx264", audio_codec="copy", fps=RENDER_FPS,702                logger='bar', threads=os.cpu_count(), preset='ultrafast'703            )704            print("Audio stream successfully copied!")705        except Exception:706            # Fallback to AAC encoding if copy fails707            print("Direct audio copy failed, falling back to high-quality AAC encoding...")708            final_clip.write_videofile(709                temp_fps1_path, codec="libx264", audio_codec="aac",710                audio_bitrate="320k", fps=RENDER_FPS,711                logger='bar', threads=os.cpu_count(), preset='ultrafast')712            print("High-quality AAC audio encoding complete.")713 714        final_clip.close()715        716        # Step 2: Use FFmpeg to quickly increase the framerate to 24 FPS717        print(f"\nStep 2/2: Remuxing video to {PLAYBACK_FPS} FPS...")718 719        # --- Use fractional progress (current/total) ---720        progress(4 / TOTAL_STEPS, desc=f"Stage 5/{TOTAL_STEPS}: Finalizing Video")721        722        # --- Finalizing ---723        increase_video_framerate(temp_fps1_path, final_output_path, target_fps=PLAYBACK_FPS)724        725        return final_output_path726        727    except Exception as e:728        # Re-raise the exception to be caught and displayed by Gradio729        raise e730    finally:731        # Step 3: Clean up the temporary file regardless of success or failure732        for f in [temp_fps1_path, temp_audio_path]:733            if os.path.exists(f):734                print(f"Cleaning up temporary file: {f}")735                os.remove(f)736 737# --- Gradio UI ---738with gr.Blocks(title="Spectrogram Video Generator") as iface:739    gr.Markdown("# Spectrogram Video Generator")740    with gr.Row():741        with gr.Column(scale=1):742            # --- Changed to gr.Files for multi-upload ---743            audio_inputs = gr.Files(744                label="Upload Audio File(s)",745                file_count="multiple",746                file_types=["audio"]747            )748            749            # --- Grouped Image Section ---750            with gr.Accordion("Grouped Image Backgrounds (Advanced)", open=False):751                gr.Markdown("Define groups of tracks and assign specific images to them. Tracks are numbered globally starting from 1 across all uploaded files.")752                753                MAX_GROUPS = 10754                group_track_inputs = []755                group_image_inputs = []756                group_accordions = []757 758                # --- Create a centralized update function ---759                def update_group_visibility(target_count: int):760                    """Updates the visibility of all group accordions and the state of the control buttons."""761                    # Clamp the target count to be within bounds762                    target_count = max(1, min(target_count, MAX_GROUPS))763                    764                    updates = {visible_groups_state: target_count}765                    # Update visibility for each accordion766                    for i in range(MAX_GROUPS):767                        updates[group_accordions[i]] = gr.update(visible=(i < target_count))768                    769                    # Update button states770                    updates[add_group_btn] = gr.update(visible=(target_count < MAX_GROUPS))771                    updates[remove_group_btn] = gr.update(interactive=(target_count > 1))772                    773                    return updates774                775                # --- Create simple wrapper functions for adding and removing ---776                def add_group(current_count: int):777                    return update_group_visibility(current_count + 1)778                779                def remove_group(current_count: int):780                    return update_group_visibility(current_count - 1)781 782                # Pre-build all group components783                for i in range(MAX_GROUPS):784                    with gr.Accordion(f"Image Group {i+1}", open=False, visible=(i==0)) as acc:785                        track_input = gr.Textbox(label=f"Tracks for Group {i+1} (e.g., '1-4, 7')")786                        image_input = gr.Files(label=f"Images for Group {i+1}", file_count="multiple", file_types=[".png", ".jpg", ".jpeg", ".webp", ".avif"])787                        group_track_inputs.append(track_input)788                        group_image_inputs.append(image_input)789                        group_accordions.append(acc)790                        791                visible_groups_state = gr.State(1)792                # --- Add a remove button and put both in a row ---793                with gr.Row():794                    remove_group_btn = gr.Button("- Remove Last Group", variant="secondary", interactive=False)795                    add_group_btn = gr.Button("+ Add Image Group", variant="secondary")796                797                with gr.Accordion("Fallback / Default Images", open=True):798                    gr.Markdown("These images will be used for any tracks not assigned to a specific group above.")799                    fallback_image_input = gr.Files(label="Fallback Images", file_count="multiple", file_types=[".png", ".jpg", ".jpeg", ".webp", ".avif"])800            801            # --- Renamed for clarity ---802            with gr.Accordion("General Visualizer Options", open=True):803                with gr.Row():804                    width_input = gr.Number(value=1920, label="Video Width (px)", precision=0)805                    height_input = gr.Number(value=1080, label="Video Height (px)", precision=0)806                fg_color = gr.ColorPicker(value="#71808c", label="Spectrogram Bar Color")807                bg_color = gr.ColorPicker(value="#2C3E50", label="Background Color (if no images)")808 809            # --- Dedicated Accordion for Spectrogram Bar Style ---810            with gr.Accordion("Spectrogram Bar Style", open=True):811                n_bands_slider = gr.Slider(minimum=8, maximum=256, value=64, step=1, label="Number of Spectrogram Bars")812                bar_spacing_slider = gr.Slider(minimum=0, maximum=10, value=2, step=1, label="Bar/Block Spacing (px)")813                814                # --- Replaced Checkbox with Radio for mirror modes ---815                mirror_mode_radio = gr.Radio(816                    choices=["Off", "Horizontal (Top/Bottom)", "Vertical (Left/Right)"],817                    value="Off",818                    label="Symmetry / Mirror Mode"819                )820                821                with gr.Row():822                    bar_style_radio = gr.Radio(823                        choices=["Solid Bars", "Stacked Blocks"],824                        value="Solid Bars",825                        label="Bar Style"826                    )827                    num_blocks_slider = gr.Slider(828                        minimum=5, maximum=50, value=20, step=1, 829                        label="Number of Blocks per Bar",830                        visible=False # Initially hidden831                    )832                833                # --- Function to dynamically show/hide the block count slider ---834                def update_block_slider_visibility(bar_style):835                    return gr.update(visible=(bar_style == "Stacked Blocks"))836                837                bar_style_radio.change(838                    fn=update_block_slider_visibility,839                    inputs=bar_style_radio,840                    outputs=num_blocks_slider841                )842            843            with gr.Accordion("Text Overlay Options", open=True):844                gr.Markdown(845                    "**Note:** The title overlay feature automatically detects if a file has an embedded CUE sheet. If not, the filename will be used as the title."846                )847                gr.Markdown("---")848                # --- Checkbox for number formatting ---849                format_double_digits_checkbox = gr.Checkbox(label="Format track numbers as double digits (e.g., 01, 05-09)", value=True)850                gr.Markdown("If the CUE sheet or filenames contain non-English characters, please select a compatible font.")851 852                # Define a priority list for default fonts, starting with common Japanese ones.853                # This list can include multiple names for the same font to improve matching.854                preferred_fonts = [855                    "Yu Gothic", "游ゴシック",856                    "MS Gothic", "MS ゴシック",857                    "Meiryo", "メイリオ",858                    "Hiragino Kaku Gothic ProN", # Common on macOS859                    "Microsoft JhengHei", # Fallback to Traditional Chinese860                    "Arial" # Generic fallback861                ]862                default_font = None863                # Find the first available font from the preferred list864                for font in preferred_fonts:865                    for candidate in FONT_DISPLAY_NAMES:866                        if candidate.startswith(font) or font in candidate:867                            default_font = candidate868                            break869                    if default_font:870                        break871                    872                # If none of the preferred fonts are found, use the first available font as a last resort873                if not default_font and FONT_DISPLAY_NAMES:874                    default_font = FONT_DISPLAY_NAMES[0]875 876                font_name_dd = gr.Dropdown(choices=FONT_DISPLAY_NAMES, value=default_font, label="Font Family")877 878                with gr.Row():879                    font_size_slider = gr.Slider(minimum=12, maximum=256, value=80, step=1, label="Font Size")880                    font_color_picker = gr.ColorPicker(value="#FFFFFF", label="Font Color")881 882                with gr.Row():883                    font_bg_color_picker = gr.ColorPicker(value="#000000", label="Text BG Color")884                    font_bg_alpha_slider = gr.Slider(minimum=0.0, maximum=1.0, value=0.6, step=0.05, label="Text BG Opacity")885                    886                gr.Markdown("Text Position")887                with gr.Row():888                    pos_h_radio = gr.Radio(["left", "center", "right"], value="center", label="Horizontal Align")889                    pos_v_radio = gr.Radio(["top", "center", "bottom"], value="bottom", label="Vertical Align")890            891            submit_btn = gr.Button("Generate Video", variant="primary")892            893        with gr.Column(scale=2):894            video_output = gr.Video(label="Generated Video")895       896    # --- Define the full list of outputs for the update functions ---897    group_update_outputs = [visible_groups_state, add_group_btn, remove_group_btn] + group_accordions898 899    # Connect the "Add Group" button to its update function900    add_group_btn.click(901        fn=add_group,902        inputs=visible_groups_state,903        outputs=group_update_outputs904    )905 906    remove_group_btn.click(907        fn=remove_group,908        inputs=visible_groups_state,909        outputs=group_update_outputs910    )911    912    # --- Define the master list of all inputs for the main button ---913    all_inputs = [audio_inputs] + group_track_inputs + group_image_inputs + [914        fallback_image_input,915        format_double_digits_checkbox,916        width_input, height_input,917        fg_color, bg_color, 918        # --- Add spectrogram style inputs in correct order ---919        n_bands_slider, bar_spacing_slider, mirror_mode_radio,920        bar_style_radio, num_blocks_slider,921        # --- Text and font inputs ---922        font_name_dd, font_size_slider, font_color_picker,923        font_bg_color_picker, font_bg_alpha_slider,924        pos_h_radio, pos_v_radio925    ]926 927    submit_btn.click(928        fn=process_audio_to_video,929        inputs=all_inputs,930        outputs=video_output,931        show_progress="full"932    )933 934if __name__ == "__main__":935    iface.launch(inbrowser=True)