avans06/Audio_Spectrogram_Video_Generator
1
1import gradio as gr2import librosa3import numpy as np4import re5import os6import time7import struct8import subprocess9import soundfile as sf10import matplotlib.font_manager as fm11from PIL import ImageFont12from typing import Tuple, List, Dict, Set13from mutagen.flac import FLAC14from moviepy import CompositeVideoClip, TextClip, VideoClip, AudioFileClip, ImageClip15 16# --- Font Scanning and Management ---17def get_font_display_name(font_path: str) -> Tuple[str, str]:18 """19 A robust TTF/TTC parser based on the user's final design.20 It reads the 'name' table to find the localized "Full Font Name" (nameID=4).21 Returns a tuple of (display_name, language_tag {'zh'/'ja'/'ko'/'en'/'other'}).22 """23 def decode_name_string(name_bytes: bytes, platform_id: int, encoding_id: int) -> str:24 """Decodes the name string based on platform and encoding IDs."""25 try:26 if platform_id == 3 and encoding_id in [1, 10]: # Windows, Unicode27 return name_bytes.decode('utf_16_be').strip('\x00')28 elif platform_id == 1 and encoding_id == 0: # Macintosh, Roman29 return name_bytes.decode('mac_roman').strip('\x00')30 elif platform_id == 0: # Unicode31 return name_bytes.decode('utf_16_be').strip('\x00')32 else: # Fallback33 return name_bytes.decode('utf_8', errors='ignore').strip('\x00')34 except Exception:35 return None36 37 try:38 with open(font_path, 'rb') as f: data = f.read()39 def read_ushort(offset):40 return struct.unpack('>H', data[offset:offset+2])[0]41 def read_ulong(offset):42 return struct.unpack('>I', data[offset:offset+4])[0]43 font_offsets = [0]44 # Check for TTC (TrueType Collection) header45 if data[:4] == b'ttcf':46 num_fonts = read_ulong(8)47 font_offsets = [read_ulong(12 + i * 4) for i in range(num_fonts)]48 49 # For simplicity, we only parse the first font in a TTC50 font_offset = font_offsets[0]51 52 num_tables = read_ushort(font_offset + 4)53 name_table_offset = -154 # Locate the 'name' table55 for i in range(num_tables):56 entry_offset = font_offset + 12 + i * 1657 tag = data[entry_offset:entry_offset+4]58 if tag == b'name':59 name_table_offset = read_ulong(entry_offset + 8)60 break61 62 if name_table_offset == -1:63 return None, None64 65 count, string_offset = read_ushort(name_table_offset + 2), read_ushort(name_table_offset + 4)66 name_candidates = {}67 # Iterate through all name records68 for i in range(count):69 rec_offset = name_table_offset + 6 + i * 1270 platform_id, encoding_id, language_id, name_id, length, offset = struct.unpack('>HHHHHH', data[rec_offset:rec_offset+12])71 72 if name_id == 4: # We only care about the "Full Font Name"73 string_pos = name_table_offset + string_offset + offset74 value = decode_name_string(data[string_pos : string_pos + length], platform_id, encoding_id)75 76 if value:77 # Store candidates based on language ID78 if language_id in [1028, 2052, 3076, 4100, 5124]:79 name_candidates["zh"] = value80 elif language_id == 1041:81 name_candidates["ja"] = value82 elif language_id == 1042:83 name_candidates["ko"] = value84 elif language_id in [1033, 0]:85 name_candidates["en"] = value86 else:87 if "other" not in name_candidates:88 name_candidates["other"] = value89 90 # Return the best candidate based on language priority91 if name_candidates.get("zh"):92 return name_candidates.get("zh"), "zh"93 if name_candidates.get("ja"):94 return name_candidates.get("ja"), "ja"95 if name_candidates.get("ko"):96 return name_candidates.get("ko"), "ko"97 if name_candidates.get("other"):98 return name_candidates.get("other"), "other"99 if name_candidates.get("en"):100 return name_candidates.get("en"), "en"101 return None, None102 103 except Exception:104 return None, None105 106def get_font_data() -> Tuple[Dict[str, str], List[str]]:107 """108 Scans system fonts, parses their display names, and returns a sorted list109 with a corresponding name-to-path map.110 """111 font_map = {}112 found_names = [] # Stores (display_name, is_fallback, lang_tag)113 114 # Scan for both .ttf and .ttc files115 ttf_files = fm.findSystemFonts(fontpaths=None, fontext='ttf')116 ttc_files = fm.findSystemFonts(fontpaths=None, fontext='ttc')117 all_font_files = list(set(ttf_files + ttc_files))118 119 for path in all_font_files:120 display_name, lang_tag = get_font_display_name(path)121 is_fallback = display_name is None122 123 if is_fallback:124 # Create a fallback name from the filename125 display_name = os.path.splitext(os.path.basename(path))[0].replace('-', ' ').replace('_', ' ').title()126 lang_tag = 'fallback'127 128 if display_name and display_name not in font_map:129 font_map[display_name] = path130 found_names.append((display_name, is_fallback, lang_tag))131 132 # Define sort priority for languages133 sort_order = {'zh': 0, 'ja': 1, 'ko': 2, 'en': 3, 'other': 4, 'fallback': 5}134 135 # Sort by priority, then alphabetically136 found_names.sort(key=lambda x: (sort_order.get(x[2], 99), x[0]))137 138 sorted_display_names = [name for name, _, _ in found_names]139 return font_map, sorted_display_names140 141print("Scanning system fonts and parsing names...")142SYSTEM_FONTS_MAP, FONT_DISPLAY_NAMES = get_font_data()143print(f"Scan complete. Found {len(FONT_DISPLAY_NAMES)} available fonts.")144 145 146# --- CUE Sheet Parsing Logic ---147def cue_time_to_seconds(time_str: str) -> float:148 try:149 minutes, seconds, frames = map(int, time_str.split(':'))150 return minutes * 60 + seconds + frames / 75.0151 except ValueError:152 return 0.0153 154def parse_cue_sheet_manually(cue_data: str) -> List[Dict[str, any]]:155 tracks = []156 current_track_info = None157 for line in cue_data.splitlines():158 line = line.strip()159 if line.upper().startswith('TRACK'):160 if current_track_info and 'title' in current_track_info and 'start_time' in current_track_info:161 tracks.append(current_track_info)162 current_track_info = {}163 continue164 if current_track_info is not None:165 title_match = re.search(r'TITLE\s+"(.*?)"', line, re.IGNORECASE)166 if title_match:167 current_track_info['title'] = title_match.group(1)168 continue169 index_match = re.search(r'INDEX\s+01\s+(\d{2}:\d{2}:\d{2})', line, re.IGNORECASE)170 if index_match:171 current_track_info['start_time'] = cue_time_to_seconds(index_match.group(1))172 continue173 if current_track_info and 'title' in current_track_info and 'start_time' in current_track_info:174 tracks.append(current_track_info)175 return tracks176 177 178# --- FFmpeg Framerate Conversion ---179def increase_video_framerate(input_path: str, output_path: str, target_fps: int = 24):180 """181 Uses FFmpeg to increase the video's framerate without re-encoding.182 This is extremely fast as it only copies streams and changes metadata.183 184 Args:185 input_path (str): Path to the low-framerate video file.186 output_path (str): Path for the final, high-framerate video file.187 target_fps (int): The desired output framerate.188 """189 print(f"Increasing framerate of '{input_path}' to {target_fps} FPS...")190 191 # Construct the FFmpeg command based on the user's specification192 command = [193 'ffmpeg',194 '-y', # Overwrite output file if exists195 '-i', input_path,196 '-map', '0', # Map all streams (video, audio, subtitles)197 '-vf', f'fps={target_fps}', # Use fps filter to convert framerate to 24198 '-c:v', 'libx264', # Re-encode video with H.264 codec199 '-preset', 'fast', # Encoding speed/quality tradeoff200 '-crf', '18', # Quality (lower is better)201 '-c:a', 'copy', # Copy audio without re-encoding202 output_path203 ]204 205 try:206 # Execute the command207 # Using capture_output to hide ffmpeg logs from the main console unless an error occurs208 result = subprocess.run(command, check=True, capture_output=True, text=True)209 print("Framerate increase successful.")210 except FileNotFoundError:211 # This error occurs if FFmpeg is not installed or not in the system's PATH212 raise gr.Error("FFmpeg not found. Please ensure FFmpeg is installed and accessible in your system's PATH.")213 except subprocess.CalledProcessError as e:214 # This error occurs if FFmpeg returns a non-zero exit code215 print("FFmpeg error output:\n", e.stderr)216 raise gr.Error(f"FFmpeg failed to increase the framerate. See console for details. Error: {e.stderr}")217 218 219# --- HELPER FUNCTION for parsing track ranges ---220def parse_track_ranges(range_str: str) -> Set[int]:221 """Parses a string like '1-4, 7, 10-13' into a set of integers."""222 if not range_str:223 return set()224 225 indices = set()226 parts = range_str.split(',')227 for part in parts:228 part = part.strip()229 if not part:230 continue231 if '-' in part:232 try:233 start, end = map(int, part.split('-'))234 indices.update(range(start, end + 1))235 except ValueError:236 print(f"Warning: Could not parse range '{part}'. Skipping.")237 else:238 try:239 indices.add(int(part))240 except ValueError:241 print(f"Warning: Could not parse track number '{part}'. Skipping.")242 return indices243 244 245# --- Main Processing Function ---246def process_audio_to_video(*args, progress=gr.Progress(track_tqdm=True)):247 # --- Correctly unpack all arguments from *args using slicing ---248 MAX_GROUPS = 10 # This MUST match the UI definition249 250 # Define the structure of the *args tuple based on the `all_inputs` list251 audio_files = args[0]252 253 # Slice the args tuple to get the continuous blocks of inputs254 all_track_strs = args[1 : 1 + MAX_GROUPS]255 all_image_lists = args[1 + MAX_GROUPS : 1 + MAX_GROUPS * 2]256 257 # Group inputs are packed in pairs (track_str, image_list)258 group_definitions = []259 for i in range(MAX_GROUPS):260 group_definitions.append({261 "tracks_str": all_track_strs[i],262 "images": all_image_lists[i]263 })264 265 # Unpack the remaining arguments with correct indexing266 arg_offset = 1 + MAX_GROUPS * 2267 fallback_images = args[arg_offset]268 format_double_digits = args[arg_offset + 1]269 video_width = args[arg_offset + 2]270 video_height = args[arg_offset + 3]271 spec_fg_color = args[arg_offset + 4]272 spec_bg_color = args[arg_offset + 5]273 274 # --- NEW: Unpack spectrogram style arguments ---275 n_bands = int(args[arg_offset + 6])276 bar_spacing = int(args[arg_offset + 7])277 mirror_mode = args[arg_offset + 8] # This is now a string278 bar_style = args[arg_offset + 9]279 num_blocks = int(args[arg_offset + 10])280 281 # --- Unpack font and text arguments (indices are shifted) ---282 font_name = args[arg_offset + 11]283 font_size = args[arg_offset + 12]284 font_color = args[arg_offset + 13]285 font_bg_color = args[arg_offset + 14]286 font_bg_alpha = args[arg_offset + 15]287 pos_h = args[arg_offset + 16]288 pos_v = args[arg_offset + 17]289 290 291 if not audio_files:292 raise gr.Error("Please upload at least one audio file.")293 if not font_name:294 raise gr.Error("Please select a font from the list.")295 296 progress(0, desc="Initializing...")297 298 # Define paths for temporary and final files299 timestamp = int(time.time())300 temp_fps1_path = f"temp_{timestamp}_fps1.mp4"301 temp_audio_path = f"temp_combined_audio_{timestamp}.wav"302 final_output_path = f"final_video_{timestamp}_fps24.mp4"303 304 WIDTH, HEIGHT = int(video_width), int(video_height)305 RENDER_FPS = 1 # Render at 1 FPS306 PLAYBACK_FPS = 24 # Final playback framerate307 308 # --- A robust color parser for hex and rgb() strings ---309 def parse_color_to_rgb(color_str: str) -> Tuple[int, int, int]:310 """311 Parses a color string which can be in hex format (#RRGGBB) or312 rgb format (e.g., "rgb(255, 128, 0)").313 Returns a tuple of (R, G, B).314 """315 color_str = color_str.strip()316 if color_str.startswith('#'):317 # Handle hex format318 hex_val = color_str.lstrip('#')319 if len(hex_val) == 3: # Handle shorthand hex like #FFF320 hex_val = "".join([c*2 for c in hex_val])321 return tuple(int(hex_val[i:i+2], 16) for i in (0, 2, 4))322 elif color_str.startswith('rgb'):323 # Handle rgb format324 try:325 numbers = re.findall(r'\d+', color_str)326 return tuple(int(n) for n in numbers[:3])327 except (ValueError, IndexError):328 raise ValueError(f"Could not parse rgb color string: {color_str}")329 else:330 raise ValueError(f"Unknown color format: {color_str}")331 332 # Use the new robust parser for all color inputs333 fg_rgb, bg_rgb = parse_color_to_rgb(spec_fg_color), parse_color_to_rgb(spec_bg_color)334 grid_rgb = tuple(min(c + 40, 255) for c in bg_rgb)335 336 # Wrap the entire process in a try...finally block to ensure cleanup337 try:338 # --- Define total steps for the progress bar ---339 TOTAL_STEPS = 5340 341 # --- Stage 1: Audio Processing & Master Track List Creation ---342 master_track_list, y_accumulator, current_sr = [], [], None343 total_duration, global_track_counter = 0.0, 0344 345 # --- Use `progress.tqdm` to create a progress bar for this loop ---346 for file_idx, audio_path in enumerate(progress.tqdm(audio_files, desc=f"Stage 1/{TOTAL_STEPS}: Analyzing Audio Files")):347 # --- Load audio as stereo (or its original channel count) ---348 y, sr = librosa.load(audio_path, sr=None, mono=False)349 # If loaded audio is mono (1D array), convert it to a 2D stereo array350 # by duplicating the channel. This ensures all arrays can be concatenated.351 if y.ndim == 1:352 print(f" - Converting mono file to stereo: {os.path.basename(audio_path)}")353 y = np.stack([y, y])354 355 if current_sr is None:356 current_sr = sr357 if current_sr != sr:358 print(f"Warning: Sample rate mismatch for {os.path.basename(audio_path)}. Expected {current_sr}Hz, found {sr}Hz.")359 print(f"Resampling from {sr}Hz to {current_sr}Hz...")360 y = librosa.resample(y, orig_sr=sr, target_sr=current_sr)361 362 y_accumulator.append(y)363 # Use the first channel (y[0]) for duration calculation, which is standard practice364 file_duration = librosa.get_duration(y=y[0], sr=current_sr)365 366 # First, try to parse the CUE sheet from the audio file.367 cue_tracks = []368 if audio_path.lower().endswith('.flac'):369 try:370 audio_meta = FLAC(audio_path)371 if 'cuesheet' in audio_meta.tags:372 cue_tracks = parse_cue_sheet_manually(audio_meta.tags['cuesheet'][0])373 374 print(f"Successfully parsed {len(cue_tracks)} tracks from CUE sheet.")375 except Exception as e:376 print(f"Warning: Could not parse CUE sheet for {os.path.basename(audio_path)}: {e}")377 378 if cue_tracks:379 for track_idx, track in enumerate(cue_tracks):380 global_track_counter += 1381 start_time = track.get('start_time', 0)382 end_time = cue_tracks[track_idx+1].get('start_time', file_duration) if track_idx + 1 < len(cue_tracks) else file_duration383 master_track_list.append({"global_index": global_track_counter, "title": track.get('title', 'Unknown'), "start_time": total_duration + start_time, "end_time": total_duration + end_time})384 else:385 global_track_counter += 1386 master_track_list.append({"global_index": global_track_counter, "title": os.path.splitext(os.path.basename(audio_path))[0], "start_time": total_duration, "end_time": total_duration + file_duration})387 388 total_duration += file_duration389 390 # --- Concatenate along the time axis (axis=1) for stereo arrays ---391 y_combined = np.concatenate(y_accumulator, axis=1)392 duration = total_duration393 394 # --- Transpose the array for soundfile to write stereo correctly ---395 sf.write(temp_audio_path, y_combined.T, current_sr)396 print(f"Combined all audio files into one. Total duration: {duration:.2f}s")397 398 # --- Update progress to the next stage, use fractional progress (current/total) ---399 progress(1 / TOTAL_STEPS, desc=f"Stage 2/{TOTAL_STEPS}: Mapping Images to Tracks")400 401 # --- Stage 2: Map Tracks to Image Groups ---402 parsed_groups = [parse_track_ranges(g['tracks_str']) for g in group_definitions]403 track_to_images_map = {}404 for track_info in master_track_list:405 track_idx = track_info['global_index']406 assigned = False407 for i, group_indices in enumerate(parsed_groups):408 if track_idx in group_indices:409 track_to_images_map[track_idx] = group_definitions[i]['images']410 assigned = True411 break412 if not assigned:413 track_to_images_map[track_idx] = fallback_images414 415 # --- Stage 3: Generate ImageClips based on contiguous blocks ---416 image_clips = []417 if any(track_to_images_map.values()):418 current_track_cursor = 0419 while current_track_cursor < len(master_track_list):420 start_track_info = master_track_list[current_track_cursor]421 image_set_for_block = track_to_images_map.get(start_track_info['global_index'])422 423 # Find the end of the contiguous block of tracks that use the same image set424 end_track_cursor = current_track_cursor425 while (end_track_cursor + 1 < len(master_track_list) and426 track_to_images_map.get(master_track_list[end_track_cursor + 1]['global_index']) == image_set_for_block):427 end_track_cursor += 1428 429 end_track_info = master_track_list[end_track_cursor]430 431 block_start_time = start_track_info['start_time']432 block_end_time = end_track_info['end_time']433 block_duration = block_end_time - block_start_time434 435 if image_set_for_block and block_duration > 0:436 print(f"Creating image block for tracks {start_track_info['global_index']}-{end_track_info['global_index']} (Time: {block_start_time:.2f}s - {block_end_time:.2f}s)")437 time_per_image = block_duration / len(image_set_for_block)438 for i, img_path in enumerate(image_set_for_block):439 def create_image_layer(path, start, dur):440 try:441 img = ImageClip(path)442 scale = min(WIDTH/img.w, HEIGHT/img.h)443 resized_img = img.resized(scale)444 return CompositeVideoClip([resized_img.with_position("center")], size=(WIDTH, HEIGHT)).with_duration(dur).with_start(start)445 except Exception as e:446 print(f"Warning: Failed to process image '{path}'. Skipping. Error: {e}")447 return None448 449 clip = create_image_layer(img_path, block_start_time + i * time_per_image, time_per_image)450 if clip:451 image_clips.append(clip)452 453 current_track_cursor = end_track_cursor + 1454 455 progress(2 / TOTAL_STEPS, desc=f"Stage 3/{TOTAL_STEPS}: Generating Text & Spectrogram")456 457 # --- Stage 4: Generate Text and Spectrogram ---458 # --- Text Overlay Logic using the aggregated track info459 text_clips = [] # Text clips are now simpler as they don't depend on complex file logic anymore460 461 font_path = SYSTEM_FONTS_MAP.get(font_name)462 if not font_path:463 raise gr.Error(f"Font path for '{font_name}' not found!")464 465 # Use the robust parser for text colors as well466 font_bg_rgb = parse_color_to_rgb(font_bg_color)467 468 position = (pos_h.lower(), pos_v.lower())469 470 print(f"Using font: {font_name}, Size: {font_size}, Position: {position}")471 472 # Create the RGBA tuple for the background color.473 # The alpha value is converted from a 0.0-1.0 float to a 0-255 integer.474 bg_color_tuple = (font_bg_rgb[0], font_bg_rgb[1], font_bg_rgb[2], int(font_bg_alpha * 255))475 476 # 1. Define a maximum width for the caption. 90% of the video width is a good choice.477 caption_width = int(WIDTH * 0.9)478 479 # --- Get font metrics to calculate dynamic padding ---480 try:481 # Load the font with Pillow to access its metrics482 pil_font = ImageFont.truetype(font_path, size=font_size)483 _, descent = pil_font.getmetrics()484 # Calculate a bottom margin to compensate for the font's descent.485 # A small constant is added as a safety buffer.486 # This prevents clipping on fonts with large descenders (like 'g', 'p').487 bottom_margin = int(descent * 0.5) + 2488 print(f"Font '{font_name}' descent: {descent}. Applying dynamic bottom margin of {bottom_margin}px.")489 except Exception as e:490 # Fallback in case of any font loading error491 print(f"Warning: Could not get font metrics for '{font_name}'. Using fixed margin. Error: {e}")492 bottom_margin = int(WIDTH * 0.01) # A small fixed fallback493 494 for track in master_track_list:495 text_duration = track['end_time'] - track['start_time']496 if text_duration <= 0:497 continue498 499 # Construct display text based on pre-formatted number string500 num_str = f"{track['global_index']:02d}" if format_double_digits else str(track['global_index'])501 display_text = f"{num_str}. {track['title']}"502 503 504 # 1. Create the TextClip first without positioning to get its size505 txt_clip = TextClip(506 text=display_text.strip(),507 font_size=font_size,508 color=font_color,509 font=font_path,510 bg_color=bg_color_tuple,511 method='caption', # <-- Set method to caption512 size=(caption_width, None), # <-- Provide size for wrapping513 margin=(0, 0, 0, bottom_margin)514 ).with_position(position).with_duration(text_duration).with_start(track['start_time'])515 516 text_clips.append(txt_clip)517 518 N_FFT, HOP_LENGTH = 2048, 512519 MIN_DB, MAX_DB = -80.0, 0.0520 521 # Spectrogram calculation on combined audio522 # --- Create a mono version of audio specifically for the spectrogram ---523 # This resolves the TypeError while keeping the final audio in stereo.524 y_mono_for_spec = librosa.to_mono(y_combined)525 S_mel = librosa.feature.melspectrogram(y=y_mono_for_spec, sr=current_sr, n_fft=N_FFT, hop_length=HOP_LENGTH, n_mels=n_bands, fmax=current_sr/2)526 S_mel_db = librosa.power_to_db(S_mel, ref=np.max)527 528 # --- Pre-calculate drawing parameters for stacked block style ---529 BLOCK_SPACING = 2 # The pixel gap between stacked blocks530 if bar_style == 'Stacked Blocks':531 # Calculate the total vertical space available for the blocks themselves532 # In mirrored mode, this is based on half the screen height533 if mirror_mode == 'Vertical (Left/Right)':534 drawable_size = WIDTH // 2535 elif mirror_mode == 'Horizontal (Top/Bottom)':536 drawable_size = HEIGHT // 2537 else: # Off538 drawable_size = HEIGHT539 total_block_pixel_size = drawable_size - ((num_blocks - 1) * BLOCK_SPACING)540 # Calculate the size of a single block541 single_block_size = total_block_pixel_size / num_blocks542 543 # Frame generation logic for the spectrogram544 def frame_generator(t):545 # If images are used as background, the spectrogram's own background should be transparent.546 # Otherwise, use the selected background color.547 # Here, we will use a simple opacity setting on the final clip, so we always generate the frame.548 frame_bg = bg_rgb if not image_clips else (0,0,0) # Use black if it will be made transparent later549 frame = np.full((HEIGHT, WIDTH, 3), frame_bg, dtype=np.uint8)550 551 # Draw the grid lines only if no images are being used.552 if not image_clips:553 for i in range(1, 9):554 y_pos = int(i * (HEIGHT / 9)); frame[y_pos-1:y_pos, :] = grid_rgb555 556 # 1. Safety Check: If the spectrogram has no time frames (e.g., from an extremely short audio file),557 # return a blank frame immediately to prevent an IndexError.558 if S_mel_db.shape[1] == 0:559 return frame560 561 # 2. Use librosa.time_to_frames to accurately convert the video time `t`562 # into a spectrogram frame index. This is far more reliable than manual scaling563 # and solves the problem of missing content on the rightmost side of the video.564 time_idx = librosa.time_to_frames(t, sr=current_sr, hop_length=HOP_LENGTH)565 566 # 3. Boundary Protection: Although time_to_frames is accurate, this extra `min`567 # call acts as a safeguard to ensure the index never exceeds the array's568 # maximum valid index, preventing any edge-case errors.569 time_idx = min(time_idx, S_mel_db.shape[1] - 1)570 571 # --- RENDER LOGIC FOR VERTICAL MIRROR ---572 if mirror_mode == 'Vertical (Left/Right)':573 center_x = WIDTH // 2574 max_pixel_length = WIDTH // 2575 bar_height = HEIGHT / n_bands576 577 for i in range(n_bands):578 energy_db = S_mel_db[i, time_idx]579 norm_height = np.clip((energy_db - MIN_DB) / (MAX_DB - MIN_DB), 0, 1)580 if norm_height == 0:581 continue582 583 # --- Calculate y-coords from bottom-to-top ---584 # This makes low frequencies appear at the bottom and high frequencies at the top.585 y_start = int(HEIGHT - (i + 1) * bar_height)586 y_end = int(HEIGHT - i * bar_height)587 588 # Apply spacing to create a gap above the current bar.589 y_start_with_spacing = y_start + bar_spacing590 591 # Ensure the bar still has visible height after spacing592 if y_start_with_spacing >= y_end:593 continue594 595 if bar_style == 'Stacked Blocks':596 blocks_to_draw = int(norm_height * num_blocks)597 if blocks_to_draw == 0:598 continue599 600 for j in range(blocks_to_draw):601 block_left_x = center_x + (j * (single_block_size + BLOCK_SPACING))602 block_right_x = block_left_x + single_block_size603 # Draw right side604 frame[y_start_with_spacing:y_end, int(block_left_x):int(block_right_x)] = fg_rgb605 # Draw mirrored left side606 frame[y_start_with_spacing:y_end, int(center_x - (block_right_x - center_x)):int(center_x - (block_left_x - center_x))] = fg_rgb607 else: # Solid Bars608 bar_pixel_length = int(norm_height * max_pixel_length)609 if bar_pixel_length < 1:610 continue611 612 # Draw right side613 frame[y_start_with_spacing:y_end, center_x : center_x + bar_pixel_length] = fg_rgb614 # Draw mirrored left side615 frame[y_start_with_spacing:y_end, center_x - bar_pixel_length : center_x] = fg_rgb616 617 # --- RENDER LOGIC FOR HORIZONTAL MIRROR AND OFF ---618 else:619 bar_width = WIDTH / n_bands620 is_horizontal_mirror = (mirror_mode == 'Horizontal (Top/Bottom)')621 622 # Determine rendering parameters based on whether the view is mirrored623 if is_horizontal_mirror:624 center_y = HEIGHT // 2625 max_pixel_height = HEIGHT // 2626 else: # Off627 center_y = HEIGHT # The "center" is the bottom of the screen628 max_pixel_height = HEIGHT629 630 # Loop through each frequency band to draw its bar/blocks631 for i in range(n_bands):632 energy_db = S_mel_db[i, time_idx]633 634 # The denominator should be the range of DB values (MAX_DB - MIN_DB).635 # Since MAX_DB is 0, this simplifies to -MIN_DB, which is a positive 80.0.636 # This prevents the division by zero warning.637 norm_height = np.clip((energy_db - MIN_DB) / (MAX_DB - MIN_DB), 0, 1)638 639 if norm_height == 0:640 continue641 642 # Calculate the horizontal position of the current bar643 x_start = int(i * bar_width)644 x_end = int((i + 1) * bar_width - bar_spacing)645 646 # --- Main rendering logic: switches between styles ---647 if bar_style == 'Stacked Blocks':648 # Calculate how many blocks to draw based on energy649 blocks_to_draw = int(norm_height * num_blocks)650 if blocks_to_draw == 0:651 continue652 653 # Draw each block from the bottom up654 for j in range(blocks_to_draw):655 # Calculate the Y coordinates for this specific block656 block_bottom_y = center_y - (j * (single_block_size + BLOCK_SPACING))657 block_top_y = block_bottom_y - single_block_size658 frame[int(block_top_y):int(block_bottom_y), x_start:x_end] = fg_rgb659 660 if is_horizontal_mirror:661 frame[int(center_y + (center_y - block_bottom_y)):int(center_y + (center_y - block_top_y)), x_start:x_end] = fg_rgb662 else: # Solid Bars663 # Calculate the total height of the solid bar664 bar_pixel_height = int(norm_height * max_pixel_height)665 666 if bar_pixel_height < 1:667 continue668 669 frame[center_y - bar_pixel_height : center_y, x_start:x_end] = fg_rgb670 671 if is_horizontal_mirror:672 frame[center_y : center_y + bar_pixel_height, x_start:x_end] = fg_rgb673 return frame674 675 video_clip = VideoClip(frame_function=frame_generator, duration=duration)676 677 # --- Set Spectrogram Opacity ---678 # If image clips were created, make the spectrogram layer 50% transparent.679 if image_clips:680 print("Applying 50% opacity to spectrogram layer.")681 video_clip = video_clip.with_opacity(0.5)682 683 # --- Use fractional progress (current/total) ---684 progress(3 / TOTAL_STEPS, desc=f"Stage 4/{TOTAL_STEPS}: Rendering Base Video")685 686 # --- Composition and Rendering ---687 audio_clip = AudioFileClip(temp_audio_path)688 689 # --- Clip Composition ---690 # The final composition order is important: images at the bottom, then spectrogram, then text.691 # The base layer is now the list of image clips.692 final_layers = image_clips + [video_clip] + text_clips693 final_clip = CompositeVideoClip(final_layers, size=(WIDTH, HEIGHT)).with_audio(audio_clip)694 695 # Step 1: Render the slow, 1 FPS intermediate file696 print(f"Step 1/2: Rendering base video at {RENDER_FPS} FPS...")697 try:698 # Attempt to copy audio stream directly699 print("Attempting to copy audio stream directly...")700 final_clip.write_videofile(701 temp_fps1_path, codec="libx264", audio_codec="copy", fps=RENDER_FPS,702 logger='bar', threads=os.cpu_count(), preset='ultrafast'703 )704 print("Audio stream successfully copied!")705 except Exception:706 # Fallback to AAC encoding if copy fails707 print("Direct audio copy failed, falling back to high-quality AAC encoding...")708 final_clip.write_videofile(709 temp_fps1_path, codec="libx264", audio_codec="aac",710 audio_bitrate="320k", fps=RENDER_FPS,711 logger='bar', threads=os.cpu_count(), preset='ultrafast')712 print("High-quality AAC audio encoding complete.")713 714 final_clip.close()715 716 # Step 2: Use FFmpeg to quickly increase the framerate to 24 FPS717 print(f"\nStep 2/2: Remuxing video to {PLAYBACK_FPS} FPS...")718 719 # --- Use fractional progress (current/total) ---720 progress(4 / TOTAL_STEPS, desc=f"Stage 5/{TOTAL_STEPS}: Finalizing Video")721 722 # --- Finalizing ---723 increase_video_framerate(temp_fps1_path, final_output_path, target_fps=PLAYBACK_FPS)724 725 return final_output_path726 727 except Exception as e:728 # Re-raise the exception to be caught and displayed by Gradio729 raise e730 finally:731 # Step 3: Clean up the temporary file regardless of success or failure732 for f in [temp_fps1_path, temp_audio_path]:733 if os.path.exists(f):734 print(f"Cleaning up temporary file: {f}")735 os.remove(f)736 737# --- Gradio UI ---738with gr.Blocks(title="Spectrogram Video Generator") as iface:739 gr.Markdown("# Spectrogram Video Generator")740 with gr.Row():741 with gr.Column(scale=1):742 # --- Changed to gr.Files for multi-upload ---743 audio_inputs = gr.Files(744 label="Upload Audio File(s)",745 file_count="multiple",746 file_types=["audio"]747 )748 749 # --- Grouped Image Section ---750 with gr.Accordion("Grouped Image Backgrounds (Advanced)", open=False):751 gr.Markdown("Define groups of tracks and assign specific images to them. Tracks are numbered globally starting from 1 across all uploaded files.")752 753 MAX_GROUPS = 10754 group_track_inputs = []755 group_image_inputs = []756 group_accordions = []757 758 # --- Create a centralized update function ---759 def update_group_visibility(target_count: int):760 """Updates the visibility of all group accordions and the state of the control buttons."""761 # Clamp the target count to be within bounds762 target_count = max(1, min(target_count, MAX_GROUPS))763 764 updates = {visible_groups_state: target_count}765 # Update visibility for each accordion766 for i in range(MAX_GROUPS):767 updates[group_accordions[i]] = gr.update(visible=(i < target_count))768 769 # Update button states770 updates[add_group_btn] = gr.update(visible=(target_count < MAX_GROUPS))771 updates[remove_group_btn] = gr.update(interactive=(target_count > 1))772 773 return updates774 775 # --- Create simple wrapper functions for adding and removing ---776 def add_group(current_count: int):777 return update_group_visibility(current_count + 1)778 779 def remove_group(current_count: int):780 return update_group_visibility(current_count - 1)781 782 # Pre-build all group components783 for i in range(MAX_GROUPS):784 with gr.Accordion(f"Image Group {i+1}", open=False, visible=(i==0)) as acc:785 track_input = gr.Textbox(label=f"Tracks for Group {i+1} (e.g., '1-4, 7')")786 image_input = gr.Files(label=f"Images for Group {i+1}", file_count="multiple", file_types=[".png", ".jpg", ".jpeg", ".webp", ".avif"])787 group_track_inputs.append(track_input)788 group_image_inputs.append(image_input)789 group_accordions.append(acc)790 791 visible_groups_state = gr.State(1)792 # --- Add a remove button and put both in a row ---793 with gr.Row():794 remove_group_btn = gr.Button("- Remove Last Group", variant="secondary", interactive=False)795 add_group_btn = gr.Button("+ Add Image Group", variant="secondary")796 797 with gr.Accordion("Fallback / Default Images", open=True):798 gr.Markdown("These images will be used for any tracks not assigned to a specific group above.")799 fallback_image_input = gr.Files(label="Fallback Images", file_count="multiple", file_types=[".png", ".jpg", ".jpeg", ".webp", ".avif"])800 801 # --- Renamed for clarity ---802 with gr.Accordion("General Visualizer Options", open=True):803 with gr.Row():804 width_input = gr.Number(value=1920, label="Video Width (px)", precision=0)805 height_input = gr.Number(value=1080, label="Video Height (px)", precision=0)806 fg_color = gr.ColorPicker(value="#71808c", label="Spectrogram Bar Color")807 bg_color = gr.ColorPicker(value="#2C3E50", label="Background Color (if no images)")808 809 # --- Dedicated Accordion for Spectrogram Bar Style ---810 with gr.Accordion("Spectrogram Bar Style", open=True):811 n_bands_slider = gr.Slider(minimum=8, maximum=256, value=64, step=1, label="Number of Spectrogram Bars")812 bar_spacing_slider = gr.Slider(minimum=0, maximum=10, value=2, step=1, label="Bar/Block Spacing (px)")813 814 # --- Replaced Checkbox with Radio for mirror modes ---815 mirror_mode_radio = gr.Radio(816 choices=["Off", "Horizontal (Top/Bottom)", "Vertical (Left/Right)"],817 value="Off",818 label="Symmetry / Mirror Mode"819 )820 821 with gr.Row():822 bar_style_radio = gr.Radio(823 choices=["Solid Bars", "Stacked Blocks"],824 value="Solid Bars",825 label="Bar Style"826 )827 num_blocks_slider = gr.Slider(828 minimum=5, maximum=50, value=20, step=1, 829 label="Number of Blocks per Bar",830 visible=False # Initially hidden831 )832 833 # --- Function to dynamically show/hide the block count slider ---834 def update_block_slider_visibility(bar_style):835 return gr.update(visible=(bar_style == "Stacked Blocks"))836 837 bar_style_radio.change(838 fn=update_block_slider_visibility,839 inputs=bar_style_radio,840 outputs=num_blocks_slider841 )842 843 with gr.Accordion("Text Overlay Options", open=True):844 gr.Markdown(845 "**Note:** The title overlay feature automatically detects if a file has an embedded CUE sheet. If not, the filename will be used as the title."846 )847 gr.Markdown("---")848 # --- Checkbox for number formatting ---849 format_double_digits_checkbox = gr.Checkbox(label="Format track numbers as double digits (e.g., 01, 05-09)", value=True)850 gr.Markdown("If the CUE sheet or filenames contain non-English characters, please select a compatible font.")851 852 # Define a priority list for default fonts, starting with common Japanese ones.853 # This list can include multiple names for the same font to improve matching.854 preferred_fonts = [855 "Yu Gothic", "游ゴシック",856 "MS Gothic", "MS ゴシック",857 "Meiryo", "メイリオ",858 "Hiragino Kaku Gothic ProN", # Common on macOS859 "Microsoft JhengHei", # Fallback to Traditional Chinese860 "Arial" # Generic fallback861 ]862 default_font = None863 # Find the first available font from the preferred list864 for font in preferred_fonts:865 for candidate in FONT_DISPLAY_NAMES:866 if candidate.startswith(font) or font in candidate:867 default_font = candidate868 break869 if default_font:870 break871 872 # If none of the preferred fonts are found, use the first available font as a last resort873 if not default_font and FONT_DISPLAY_NAMES:874 default_font = FONT_DISPLAY_NAMES[0]875 876 font_name_dd = gr.Dropdown(choices=FONT_DISPLAY_NAMES, value=default_font, label="Font Family")877 878 with gr.Row():879 font_size_slider = gr.Slider(minimum=12, maximum=256, value=80, step=1, label="Font Size")880 font_color_picker = gr.ColorPicker(value="#FFFFFF", label="Font Color")881 882 with gr.Row():883 font_bg_color_picker = gr.ColorPicker(value="#000000", label="Text BG Color")884 font_bg_alpha_slider = gr.Slider(minimum=0.0, maximum=1.0, value=0.6, step=0.05, label="Text BG Opacity")885 886 gr.Markdown("Text Position")887 with gr.Row():888 pos_h_radio = gr.Radio(["left", "center", "right"], value="center", label="Horizontal Align")889 pos_v_radio = gr.Radio(["top", "center", "bottom"], value="bottom", label="Vertical Align")890 891 submit_btn = gr.Button("Generate Video", variant="primary")892 893 with gr.Column(scale=2):894 video_output = gr.Video(label="Generated Video")895 896 # --- Define the full list of outputs for the update functions ---897 group_update_outputs = [visible_groups_state, add_group_btn, remove_group_btn] + group_accordions898 899 # Connect the "Add Group" button to its update function900 add_group_btn.click(901 fn=add_group,902 inputs=visible_groups_state,903 outputs=group_update_outputs904 )905 906 remove_group_btn.click(907 fn=remove_group,908 inputs=visible_groups_state,909 outputs=group_update_outputs910 )911 912 # --- Define the master list of all inputs for the main button ---913 all_inputs = [audio_inputs] + group_track_inputs + group_image_inputs + [914 fallback_image_input,915 format_double_digits_checkbox,916 width_input, height_input,917 fg_color, bg_color, 918 # --- Add spectrogram style inputs in correct order ---919 n_bands_slider, bar_spacing_slider, mirror_mode_radio,920 bar_style_radio, num_blocks_slider,921 # --- Text and font inputs ---922 font_name_dd, font_size_slider, font_color_picker,923 font_bg_color_picker, font_bg_alpha_slider,924 pos_h_radio, pos_v_radio925 ]926 927 submit_btn.click(928 fn=process_audio_to_video,929 inputs=all_inputs,930 outputs=video_output,931 show_progress="full"932 )933 934if __name__ == "__main__":935 iface.launch(inbrowser=True)