Team Ai
Apppublic

PitGlobal/PDF_Layout_Inspector

sourceHugging Faceupdated 5mo agoView on Hugging Face
0likes
1from pathlib import Path2import sys3 4 5LOCAL_DEPS = Path(__file__).resolve().parent / ".deps"6 7try:8    import cv29    import numpy as np10except ImportError:11    if LOCAL_DEPS.exists():12        sys.path.insert(0, str(LOCAL_DEPS))13    import cv214    import numpy as np15 16 17CHAR_MIN_AREA = 1518CHAR_MAX_AREA = 200019CHAR_MIN_H = 420CHAR_MIN_W = 321CHAR_MIN_DENSITY = 0.0822 23 24def preprocess_text_mask_fine(image: np.ndarray) -> np.ndarray:25    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)26 27    bw = cv2.adaptiveThreshold(28        gray,29        255,30        cv2.ADAPTIVE_THRESH_GAUSSIAN_C,31        cv2.THRESH_BINARY_INV,32        31,33        15,34    )35 36    horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (60, 1))37    horiz_lines = cv2.morphologyEx(bw, cv2.MORPH_OPEN, horiz_kernel, iterations=1)38    bw = cv2.subtract(bw, horiz_lines)39 40    vert_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, 60))41    vert_lines = cv2.morphologyEx(bw, cv2.MORPH_OPEN, vert_kernel, iterations=1)42    bw = cv2.subtract(bw, vert_lines)43 44    clean_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2, 2))45    bw = cv2.morphologyEx(bw, cv2.MORPH_OPEN, clean_kernel, iterations=1)46    return bw47 48 49def clip_box(x1: int, y1: int, x2: int, y2: int, shape: tuple[int, ...]):50    h, w = shape[:2]51    x1 = max(0, min(x1, w - 1))52    y1 = max(0, min(y1, h - 1))53    x2 = max(1, min(x2, w))54    y2 = max(1, min(y2, h))55    return x1, y1, x2, y256 57 58def crop_box(img: np.ndarray, box: list[int] | tuple[int, int, int, int], pad: int = 2) -> np.ndarray:59    x1, y1, x2, y2 = box60    x1, y1, x2, y2 = clip_box(x1, y1, x2, y2, img.shape)61    x1 = max(0, x1 - pad)62    y1 = max(0, y1 - pad)63    x2 = min(img.shape[1], x2 + pad)64    y2 = min(img.shape[0], y2 + pad)65    return img[y1:y2, x1:x2].copy()66 67 68def tight_crop_binary(mask_crop: np.ndarray, margin: int = 2) -> np.ndarray:69    ys, xs = np.where(mask_crop > 0)70    if len(xs) == 0 or len(ys) == 0:71        return mask_crop.copy()72 73    x1 = max(0, xs.min() - margin)74    y1 = max(0, ys.min() - margin)75    x2 = min(mask_crop.shape[1], xs.max() + 1 + margin)76    y2 = min(mask_crop.shape[0], ys.max() + 1 + margin)77    return mask_crop[y1:y2, x1:x2].copy()78 79 80def resize_keep_center(binary_img: np.ndarray, out_h: int = 64, out_w: int = 64) -> np.ndarray:81    if binary_img.size == 0:82        return np.zeros((out_h, out_w), dtype=np.uint8)83 84    h, w = binary_img.shape[:2]85    if h == 0 or w == 0:86        return np.zeros((out_h, out_w), dtype=np.uint8)87 88    scale = min(out_w / w, out_h / h)89    new_w = max(1, int(round(w * scale)))90    new_h = max(1, int(round(h * scale)))91 92    resized = cv2.resize(binary_img, (new_w, new_h), interpolation=cv2.INTER_NEAREST)93    canvas = np.zeros((out_h, out_w), dtype=np.uint8)94 95    y0 = (out_h - new_h) // 296    x0 = (out_w - new_w) // 297    canvas[y0:y0 + new_h, x0:x0 + new_w] = resized98    return canvas99 100 101def smooth_1d(arr: np.ndarray, ksize: int = 5) -> np.ndarray:102    if arr.size == 0:103        return arr104    ksize = max(1, int(ksize))105    if ksize % 2 == 0:106        ksize += 1107    kernel = np.ones(ksize, dtype=np.float32) / ksize108    return np.convolve(arr.astype(np.float32), kernel, mode="same")109 110 111def count_active_bands_from_profile(profile: np.ndarray, thr: float, min_len: int = 2):112    active = profile >= thr113    bands = []114    start = None115 116    for i, value in enumerate(active):117        if value and start is None:118            start = i119        elif not value and start is not None:120            end = i - 1121            if end - start + 1 >= min_len:122                bands.append((start, end))123            start = None124 125    if start is not None:126        end = len(active) - 1127        if end - start + 1 >= min_len:128            bands.append((start, end))129 130    return bands131 132 133def detect_char_components(mask_crop: np.ndarray) -> list[dict[str, float]]:134    num_labels, _labels, stats, _ = cv2.connectedComponentsWithStats(mask_crop, connectivity=8)135    chars: list[dict[str, float]] = []136 137    for i in range(1, num_labels):138        x = int(stats[i, cv2.CC_STAT_LEFT])139        y = int(stats[i, cv2.CC_STAT_TOP])140        w = int(stats[i, cv2.CC_STAT_WIDTH])141        h = int(stats[i, cv2.CC_STAT_HEIGHT])142        area = int(stats[i, cv2.CC_STAT_AREA])143 144        if area < CHAR_MIN_AREA or area > CHAR_MAX_AREA:145            continue146        if h < CHAR_MIN_H or w < CHAR_MIN_W:147            continue148 149        aspect = w / max(h, 1)150        density = area / max(w * h, 1)151        if density < CHAR_MIN_DENSITY:152            continue153 154        chars.append(155            {156                "x": float(x),157                "y": float(y),158                "w": float(w),159                "h": float(h),160                "area": float(area),161                "aspect": float(aspect),162                "density": float(density),163                "cx": float(x + w / 2.0),164                "cy": float(y + h / 2.0),165            }166        )167 168    return chars169 170 171def summarize_char_geometry(chars: list[dict[str, float]], crop_h: int, crop_w: int) -> np.ndarray:172    if not chars:173        return np.zeros(14, dtype=np.float32)174 175    heights = np.array([c["h"] for c in chars], dtype=np.float32)176    widths = np.array([c["w"] for c in chars], dtype=np.float32)177    aspects = np.array([c["aspect"] for c in chars], dtype=np.float32)178    densities = np.array([c["density"] for c in chars], dtype=np.float32)179    centers_y = np.array([c["cy"] for c in chars], dtype=np.float32)180 181    median_h = float(np.median(heights))182    tall_count = float(np.sum(heights > median_h * 1.45))183    wide_count = float(np.sum(aspects > 2.5))184    top_count = float(np.sum(centers_y < crop_h / 3.0))185    mid_count = float(np.sum((centers_y >= crop_h / 3.0) & (centers_y < (2.0 * crop_h / 3.0))))186    bot_count = float(np.sum(centers_y >= (2.0 * crop_h / 3.0)))187 188    sorted_cy = np.sort(centers_y)189    max_cy_gap = float(np.max(np.diff(sorted_cy))) if sorted_cy.size >= 2 else 0.0190 191    feats = np.array(192        [193            float(len(chars)),194            float(np.mean(heights)),195            float(np.std(heights)),196            median_h,197            float(np.mean(widths)),198            float(np.mean(aspects)),199            float(np.mean(densities)),200            tall_count,201            wide_count,202            top_count,203            mid_count,204            bot_count,205            max_cy_gap,206            float(crop_h / max(crop_w, 1)),207        ],208        dtype=np.float32,209    )210    return feats211 212 213def structural_embedding(mask_crop: np.ndarray) -> tuple[np.ndarray, dict[str, float]]:214    tight = tight_crop_binary(mask_crop, margin=2)215    norm = resize_keep_center(tight, 64, 64)216    norm_f = (norm > 0).astype(np.float32)217 218    row_profile = norm_f.mean(axis=1)219    row_profile_s = smooth_1d(row_profile, 5)220    row_profile_ss = smooth_1d(row_profile, 9)221 222    peak = float(np.max(row_profile_s)) if row_profile_s.size else 0.0223    thr = max(0.06, peak * 0.40)224    bands = count_active_bands_from_profile(row_profile_s, thr=thr, min_len=2)225    band_count = float(len(bands))226 227    band_gap = 0.0228    valley_ratio = 1.0229    if len(bands) >= 2:230        first, second = bands[0], bands[1]231        band_gap = float(max(0, second[0] - first[1] - 1))232        if second[0] > first[1] + 1:233            valley = float(np.min(row_profile_s[first[1] + 1:second[0]]))234            valley_ratio = valley / max(peak, 1e-6)235 236    row_active = (row_profile_s >= thr).astype(np.float32)237    transitions = float(np.sum(np.abs(np.diff(row_active))))238    row_grad = np.abs(np.diff(row_profile_s)).astype(np.float32)239 240    h = norm_f.shape[0]241    upper = float(norm_f[: h // 3].mean())242    middle = float(norm_f[h // 3: 2 * h // 3].mean())243    lower = float(norm_f[2 * h // 3:].mean())244    mass_y = np.arange(h, dtype=np.float32)245    mass_sum = float(np.sum(row_profile) + 1e-6)246    center_y = float(np.sum(row_profile * mass_y) / mass_sum)247    spread_y = float(np.sqrt(np.sum(row_profile * ((mass_y - center_y) ** 2)) / mass_sum))248 249    chars = detect_char_components(tight)250    char_feats = summarize_char_geometry(chars, tight.shape[0], tight.shape[1])251    char_count = float(max(char_feats[0], 1.0))252    tall_ratio = float(char_feats[7] / char_count)253    wide_ratio = float(char_feats[8] / char_count)254    max_cy_gap_norm = float(char_feats[12] / max(tight.shape[0], 1))255    mean_char_h_norm = float(char_feats[1] / max(tight.shape[0], 1))256 257    # Pooling stretto sulla verticale: privilegia la forma delle bande e delle fusioni.258    vertical_map = cv2.resize(norm_f, (8, 24), interpolation=cv2.INTER_AREA).reshape(-1)259    overlap_feats = np.array(260        [261            upper,262            middle,263            lower,264            peak,265            valley_ratio,266            band_count,267            band_gap,268            transitions,269            float(norm_f.mean()),270            float(np.std(row_profile)),271            center_y / max(h - 1, 1),272            spread_y / max(h - 1, 1),273            float(np.max(row_grad)) if row_grad.size else 0.0,274            float(np.mean(row_grad)) if row_grad.size else 0.0,275            tall_ratio,276            wide_ratio,277            max_cy_gap_norm,278            mean_char_h_norm,279        ],280        dtype=np.float32,281    )282 283    overlap_vector = np.concatenate(284        [285            vertical_map.astype(np.float32),286            row_profile.astype(np.float32),287            row_profile_s.astype(np.float32),288            row_profile_ss.astype(np.float32),289            row_grad.astype(np.float32),290            overlap_feats.astype(np.float32),291            char_feats.astype(np.float32),292        ]293    ).astype(np.float32)294 295    # L'embedding finale e' volutamente specifica sulla sovrapposizione.296    embedding = np.concatenate(297        [298            overlap_vector,299        ]300    ).astype(np.float32)301 302    norm_value = np.linalg.norm(embedding)303    if norm_value > 0:304        embedding = embedding / norm_value305 306    debug = {307        "band_count": band_count,308        "band_gap": band_gap,309        "peak": peak,310        "valley_ratio": valley_ratio,311        "transitions": transitions,312        "char_count": float(char_feats[0]) if char_feats.size else 0.0,313        "tall_char_count": float(char_feats[7]) if char_feats.size else 0.0,314        "wide_char_count": float(char_feats[8]) if char_feats.size else 0.0,315        "max_cy_gap": float(char_feats[12]) if char_feats.size else 0.0,316        "tall_ratio": tall_ratio,317        "wide_ratio": wide_ratio,318        "max_cy_gap_norm": max_cy_gap_norm,319        "mean_char_h_norm": mean_char_h_norm,320        "center_y_norm": center_y / max(h - 1, 1),321        "spread_y_norm": spread_y / max(h - 1, 1),322    }323    return embedding, debug324