Team Ai
Apppublic

DevelopmentT/background-remover

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
image_processing.py161 linesDownload Raw Back to services
1"""2Image preprocessing pipeline optimised for industrial-level OCR extraction.3 4Pipeline steps:5  1. Decode raw bytes → OpenCV BGR image6  2. Grayscale conversion7  3. Smart Upscaling: Upscales low-res images to improve text detection8  4. Advanced Denoising: Non-Local Means Denoising (preserves text edges)9  5. Contrast enhancement (CLAHE)10  6. Unsharp Masking: Sharpens text edges to make them pop11  7. Adaptive Thresholding: Handles uneven lighting/shadows in scans12 13All operations use OpenCV and are designed to be production-ready.14"""15 16from __future__ import annotations17 18import logging19import cv220import numpy as np21 22from config import (23    CLAHE_CLIP_LIMIT,24    CLAHE_TILE_GRID,25)26 27logger = logging.getLogger("ocr.preprocessing")28 29 30class ImagePreprocessor:31    """Industrial-level image preprocessing pipeline for OCR."""32 33    def __init__(34        self,35        clahe_clip: float = 2.0,36        clahe_grid: tuple[int, int] = (8, 8),37        apply_threshold: bool = False, # PaddleOCR prefers grayscale gradients over harsh binary38    ) -> None:39        self._clahe = cv2.createCLAHE(clipLimit=clahe_clip, tileGridSize=clahe_grid)40        self._apply_threshold = apply_threshold41 42    # ------------------------------------------------------------------43    # Public API44    # ------------------------------------------------------------------45 46    def process(self, raw_bytes: bytes) -> np.ndarray:47        """48        Full industrial preprocessing pipeline.49 50        Args:51            raw_bytes: Raw image file bytes (PNG / JPEG / WebP / BMP / TIFF).52 53        Returns:54            Preprocessed grayscale image ready for high-accuracy OCR.55        """56        logger.info("Preprocessing started – %d bytes received", len(raw_bytes))57 58        img = self._decode(raw_bytes)59        gray = self._to_grayscale(img)60        61        # 1. Upscale low resolution images62        gray = self._upscale_if_needed(gray)63        64        # 2. Industrial Noise Removal65        denoised = self._advanced_denoise(gray)66        67        # 3. High Contrast68        enhanced = self._enhance_contrast(denoised)69        70        # 4. Text Sharpening (Unsharp Mask)71        sharpened = self._sharpen(enhanced)72 73        # 5. Optional Adaptive Thresholding74        if self._apply_threshold:75            final = self._adaptive_threshold(sharpened)76        else:77            final = sharpened78 79        logger.info(80            "Preprocessing complete – output shape %s", final.shape81        )82        return final83 84    # ------------------------------------------------------------------85    # Private steps86    # ------------------------------------------------------------------87 88    @staticmethod89    def _decode(raw_bytes: bytes) -> np.ndarray:90        """Decode raw bytes into an OpenCV BGR image."""91        buf = np.frombuffer(raw_bytes, dtype=np.uint8)92        img = cv2.imdecode(buf, cv2.IMREAD_COLOR)93        if img is None:94            raise ValueError("Failed to decode image – unsupported or corrupt file")95        return img96 97    @staticmethod98    def _to_grayscale(img: np.ndarray) -> np.ndarray:99        """Convert BGR → grayscale."""100        return cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)101 102    @staticmethod103    def _upscale_if_needed(gray: np.ndarray, min_width: int = 1200) -> np.ndarray:104        """Upscale the image using cubic interpolation if it's too small, helping OCR detect tiny text."""105        h, w = gray.shape106        if w < min_width:107            scale = min_width / w108            new_w, new_h = int(w * scale), int(h * scale)109            upscaled = cv2.resize(gray, (new_w, new_h), interpolation=cv2.INTER_CUBIC)110            logger.debug("Upscaled image from %dx%d to %dx%d", w, h, new_w, new_h)111            return upscaled112        return gray113 114    @staticmethod115    def _advanced_denoise(gray: np.ndarray) -> np.ndarray:116        """117        Non-Local Means Denoising. 118        Highly superior to Gaussian/Median blur for OCR because it removes grain/noise119        without blurring the sharp edges of text characters.120        """121        denoised = cv2.fastNlMeansDenoising(gray, None, h=10, templateWindowSize=7, searchWindowSize=21)122        logger.debug("Advanced NL-Means denoising applied")123        return denoised124 125    def _enhance_contrast(self, gray: np.ndarray) -> np.ndarray:126        """Apply CLAHE for contrast-limited adaptive histogram equalisation."""127        enhanced = self._clahe.apply(gray)128        logger.debug("CLAHE contrast enhancement applied")129        return enhanced130 131    @staticmethod132    def _sharpen(gray: np.ndarray) -> np.ndarray:133        """134        Unsharp Masking.135        Creates a slightly blurred version and subtracts it to make the edges of text highly visible.136        """137        gaussian = cv2.GaussianBlur(gray, (0, 0), 2.0)138        sharpened = cv2.addWeighted(gray, 1.5, gaussian, -0.5, 0)139        logger.debug("Unsharp masking applied")140        return sharpened141 142    @staticmethod143    def _adaptive_threshold(gray: np.ndarray) -> np.ndarray:144        """145        Adaptive Gaussian Thresholding.146        Unlike global Otsu, this calculates the threshold for small regions,147        making it perfect for scanned documents with shadows or uneven lighting.148        """149        binary = cv2.adaptiveThreshold(150            gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2151        )152        logger.debug("Adaptive thresholding applied")153        return binary154 155 156# ---------------------------------------------------------------------------157# Module-level singleton158# ---------------------------------------------------------------------------159preprocessor = ImagePreprocessor()160 161