Team Ai
Apppublic

mandada/forced-alignment-server

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
align.py62 linesDownload Raw Back to src
1import sys2import json3import whisperx4import torch5from contextlib import redirect_stdout6 7def align_audio(audio_path):8    """9    Performs transcription and forced alignment on an audio file using whisper-x.10    All informational output is sent to stderr, only the final JSON is sent to stdout.11    """12    try:13        print("[PYTHON_LOG] Alignment script started.", file=sys.stderr)14        15        # Check for GPU availability16        device = "cuda" if torch.cuda.is_available() else "cpu"17        compute_type = "float16" if device == "cuda" else "int8"18        print(f"[PYTHON_LOG] Using device: {device}, compute_type: {compute_type}", file=sys.stderr)19        20        aligned_result = None21        # Redirect stdout to stderr to capture whisperx logs without polluting the final JSON output22        with redirect_stdout(sys.stderr):23            # 1. Load whisper model24            print("[PYTHON_LOG] Loading whisper model...")25            model = whisperx.load_model("small", device, compute_type=compute_type)26            27            # 2. Load audio28            print("[PYTHON_LOG] Loading audio file...")29            audio = whisperx.load_audio(audio_path)30            31            # 3. Transcribe32            print("[PYTHON_LOG] Transcribing audio...")33            result = model.transcribe(audio, batch_size=16)34            35            # 4. Load alignment model36            print("[PYTHON_LOG] Loading alignment model...")37            model_a, metadata = whisperx.load_align_model(language_code=result["language"], device=device)38            39            # 5. Align transcription40            print("[PYTHON_LOG] Aligning transcription...")41            aligned_result = whisperx.align(result["segments"], model_a, metadata, audio, device, return_char_alignments=False)42            print("[PYTHON_LOG] Alignment complete.")43 44        # Print the final result to stdout, which is now clean45        if aligned_result:46            print(json.dumps(aligned_result))47 48    except Exception as e:49        print(f"[PYTHON_ERROR] An error occurred during alignment: {e}", file=sys.stderr)50        sys.exit(1)51 52if __name__ == "__main__":53    # The script now only needs the audio file path as an argument.54    if len(sys.argv) != 2:55        print("Usage: python align.py <audio_path>", file=sys.stderr)56        sys.exit(1)57 58    audio_file = sys.argv[1]59    60    # The text file from the original implementation is no longer needed.61    align_audio(audio_file)62