mandada/forced-alignment-server
0
1import sys2import json3import whisperx4import torch5from contextlib import redirect_stdout6 7def align_audio(audio_path):8 """9 Performs transcription and forced alignment on an audio file using whisper-x.10 All informational output is sent to stderr, only the final JSON is sent to stdout.11 """12 try:13 print("[PYTHON_LOG] Alignment script started.", file=sys.stderr)14 15 # Check for GPU availability16 device = "cuda" if torch.cuda.is_available() else "cpu"17 compute_type = "float16" if device == "cuda" else "int8"18 print(f"[PYTHON_LOG] Using device: {device}, compute_type: {compute_type}", file=sys.stderr)19 20 aligned_result = None21 # Redirect stdout to stderr to capture whisperx logs without polluting the final JSON output22 with redirect_stdout(sys.stderr):23 # 1. Load whisper model24 print("[PYTHON_LOG] Loading whisper model...")25 model = whisperx.load_model("small", device, compute_type=compute_type)26 27 # 2. Load audio28 print("[PYTHON_LOG] Loading audio file...")29 audio = whisperx.load_audio(audio_path)30 31 # 3. Transcribe32 print("[PYTHON_LOG] Transcribing audio...")33 result = model.transcribe(audio, batch_size=16)34 35 # 4. Load alignment model36 print("[PYTHON_LOG] Loading alignment model...")37 model_a, metadata = whisperx.load_align_model(language_code=result["language"], device=device)38 39 # 5. Align transcription40 print("[PYTHON_LOG] Aligning transcription...")41 aligned_result = whisperx.align(result["segments"], model_a, metadata, audio, device, return_char_alignments=False)42 print("[PYTHON_LOG] Alignment complete.")43 44 # Print the final result to stdout, which is now clean45 if aligned_result:46 print(json.dumps(aligned_result))47 48 except Exception as e:49 print(f"[PYTHON_ERROR] An error occurred during alignment: {e}", file=sys.stderr)50 sys.exit(1)51 52if __name__ == "__main__":53 # The script now only needs the audio file path as an argument.54 if len(sys.argv) != 2:55 print("Usage: python align.py <audio_path>", file=sys.stderr)56 sys.exit(1)57 58 audio_file = sys.argv[1]59 60 # The text file from the original implementation is no longer needed.61 align_audio(audio_file)62 