PitGlobal/PDF_Layout_Inspector
0
1from pathlib import Path2import sys3 4 5LOCAL_DEPS = Path(__file__).resolve().parent / ".deps"6 7try:8 import cv29 import numpy as np10except ImportError:11 if LOCAL_DEPS.exists():12 sys.path.insert(0, str(LOCAL_DEPS))13 import cv214 import numpy as np15 16 17CHAR_MIN_AREA = 1518CHAR_MAX_AREA = 200019CHAR_MIN_H = 420CHAR_MIN_W = 321CHAR_MIN_DENSITY = 0.0822 23 24def preprocess_text_mask_fine(image: np.ndarray) -> np.ndarray:25 gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)26 27 bw = cv2.adaptiveThreshold(28 gray,29 255,30 cv2.ADAPTIVE_THRESH_GAUSSIAN_C,31 cv2.THRESH_BINARY_INV,32 31,33 15,34 )35 36 horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (60, 1))37 horiz_lines = cv2.morphologyEx(bw, cv2.MORPH_OPEN, horiz_kernel, iterations=1)38 bw = cv2.subtract(bw, horiz_lines)39 40 vert_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, 60))41 vert_lines = cv2.morphologyEx(bw, cv2.MORPH_OPEN, vert_kernel, iterations=1)42 bw = cv2.subtract(bw, vert_lines)43 44 clean_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (2, 2))45 bw = cv2.morphologyEx(bw, cv2.MORPH_OPEN, clean_kernel, iterations=1)46 return bw47 48 49def clip_box(x1: int, y1: int, x2: int, y2: int, shape: tuple[int, ...]):50 h, w = shape[:2]51 x1 = max(0, min(x1, w - 1))52 y1 = max(0, min(y1, h - 1))53 x2 = max(1, min(x2, w))54 y2 = max(1, min(y2, h))55 return x1, y1, x2, y256 57 58def crop_box(img: np.ndarray, box: list[int] | tuple[int, int, int, int], pad: int = 2) -> np.ndarray:59 x1, y1, x2, y2 = box60 x1, y1, x2, y2 = clip_box(x1, y1, x2, y2, img.shape)61 x1 = max(0, x1 - pad)62 y1 = max(0, y1 - pad)63 x2 = min(img.shape[1], x2 + pad)64 y2 = min(img.shape[0], y2 + pad)65 return img[y1:y2, x1:x2].copy()66 67 68def tight_crop_binary(mask_crop: np.ndarray, margin: int = 2) -> np.ndarray:69 ys, xs = np.where(mask_crop > 0)70 if len(xs) == 0 or len(ys) == 0:71 return mask_crop.copy()72 73 x1 = max(0, xs.min() - margin)74 y1 = max(0, ys.min() - margin)75 x2 = min(mask_crop.shape[1], xs.max() + 1 + margin)76 y2 = min(mask_crop.shape[0], ys.max() + 1 + margin)77 return mask_crop[y1:y2, x1:x2].copy()78 79 80def resize_keep_center(binary_img: np.ndarray, out_h: int = 64, out_w: int = 64) -> np.ndarray:81 if binary_img.size == 0:82 return np.zeros((out_h, out_w), dtype=np.uint8)83 84 h, w = binary_img.shape[:2]85 if h == 0 or w == 0:86 return np.zeros((out_h, out_w), dtype=np.uint8)87 88 scale = min(out_w / w, out_h / h)89 new_w = max(1, int(round(w * scale)))90 new_h = max(1, int(round(h * scale)))91 92 resized = cv2.resize(binary_img, (new_w, new_h), interpolation=cv2.INTER_NEAREST)93 canvas = np.zeros((out_h, out_w), dtype=np.uint8)94 95 y0 = (out_h - new_h) // 296 x0 = (out_w - new_w) // 297 canvas[y0:y0 + new_h, x0:x0 + new_w] = resized98 return canvas99 100 101def smooth_1d(arr: np.ndarray, ksize: int = 5) -> np.ndarray:102 if arr.size == 0:103 return arr104 ksize = max(1, int(ksize))105 if ksize % 2 == 0:106 ksize += 1107 kernel = np.ones(ksize, dtype=np.float32) / ksize108 return np.convolve(arr.astype(np.float32), kernel, mode="same")109 110 111def count_active_bands_from_profile(profile: np.ndarray, thr: float, min_len: int = 2):112 active = profile >= thr113 bands = []114 start = None115 116 for i, value in enumerate(active):117 if value and start is None:118 start = i119 elif not value and start is not None:120 end = i - 1121 if end - start + 1 >= min_len:122 bands.append((start, end))123 start = None124 125 if start is not None:126 end = len(active) - 1127 if end - start + 1 >= min_len:128 bands.append((start, end))129 130 return bands131 132 133def detect_char_components(mask_crop: np.ndarray) -> list[dict[str, float]]:134 num_labels, _labels, stats, _ = cv2.connectedComponentsWithStats(mask_crop, connectivity=8)135 chars: list[dict[str, float]] = []136 137 for i in range(1, num_labels):138 x = int(stats[i, cv2.CC_STAT_LEFT])139 y = int(stats[i, cv2.CC_STAT_TOP])140 w = int(stats[i, cv2.CC_STAT_WIDTH])141 h = int(stats[i, cv2.CC_STAT_HEIGHT])142 area = int(stats[i, cv2.CC_STAT_AREA])143 144 if area < CHAR_MIN_AREA or area > CHAR_MAX_AREA:145 continue146 if h < CHAR_MIN_H or w < CHAR_MIN_W:147 continue148 149 aspect = w / max(h, 1)150 density = area / max(w * h, 1)151 if density < CHAR_MIN_DENSITY:152 continue153 154 chars.append(155 {156 "x": float(x),157 "y": float(y),158 "w": float(w),159 "h": float(h),160 "area": float(area),161 "aspect": float(aspect),162 "density": float(density),163 "cx": float(x + w / 2.0),164 "cy": float(y + h / 2.0),165 }166 )167 168 return chars169 170 171def summarize_char_geometry(chars: list[dict[str, float]], crop_h: int, crop_w: int) -> np.ndarray:172 if not chars:173 return np.zeros(14, dtype=np.float32)174 175 heights = np.array([c["h"] for c in chars], dtype=np.float32)176 widths = np.array([c["w"] for c in chars], dtype=np.float32)177 aspects = np.array([c["aspect"] for c in chars], dtype=np.float32)178 densities = np.array([c["density"] for c in chars], dtype=np.float32)179 centers_y = np.array([c["cy"] for c in chars], dtype=np.float32)180 181 median_h = float(np.median(heights))182 tall_count = float(np.sum(heights > median_h * 1.45))183 wide_count = float(np.sum(aspects > 2.5))184 top_count = float(np.sum(centers_y < crop_h / 3.0))185 mid_count = float(np.sum((centers_y >= crop_h / 3.0) & (centers_y < (2.0 * crop_h / 3.0))))186 bot_count = float(np.sum(centers_y >= (2.0 * crop_h / 3.0)))187 188 sorted_cy = np.sort(centers_y)189 max_cy_gap = float(np.max(np.diff(sorted_cy))) if sorted_cy.size >= 2 else 0.0190 191 feats = np.array(192 [193 float(len(chars)),194 float(np.mean(heights)),195 float(np.std(heights)),196 median_h,197 float(np.mean(widths)),198 float(np.mean(aspects)),199 float(np.mean(densities)),200 tall_count,201 wide_count,202 top_count,203 mid_count,204 bot_count,205 max_cy_gap,206 float(crop_h / max(crop_w, 1)),207 ],208 dtype=np.float32,209 )210 return feats211 212 213def structural_embedding(mask_crop: np.ndarray) -> tuple[np.ndarray, dict[str, float]]:214 tight = tight_crop_binary(mask_crop, margin=2)215 norm = resize_keep_center(tight, 64, 64)216 norm_f = (norm > 0).astype(np.float32)217 218 row_profile = norm_f.mean(axis=1)219 row_profile_s = smooth_1d(row_profile, 5)220 row_profile_ss = smooth_1d(row_profile, 9)221 222 peak = float(np.max(row_profile_s)) if row_profile_s.size else 0.0223 thr = max(0.06, peak * 0.40)224 bands = count_active_bands_from_profile(row_profile_s, thr=thr, min_len=2)225 band_count = float(len(bands))226 227 band_gap = 0.0228 valley_ratio = 1.0229 if len(bands) >= 2:230 first, second = bands[0], bands[1]231 band_gap = float(max(0, second[0] - first[1] - 1))232 if second[0] > first[1] + 1:233 valley = float(np.min(row_profile_s[first[1] + 1:second[0]]))234 valley_ratio = valley / max(peak, 1e-6)235 236 row_active = (row_profile_s >= thr).astype(np.float32)237 transitions = float(np.sum(np.abs(np.diff(row_active))))238 row_grad = np.abs(np.diff(row_profile_s)).astype(np.float32)239 240 h = norm_f.shape[0]241 upper = float(norm_f[: h // 3].mean())242 middle = float(norm_f[h // 3: 2 * h // 3].mean())243 lower = float(norm_f[2 * h // 3:].mean())244 mass_y = np.arange(h, dtype=np.float32)245 mass_sum = float(np.sum(row_profile) + 1e-6)246 center_y = float(np.sum(row_profile * mass_y) / mass_sum)247 spread_y = float(np.sqrt(np.sum(row_profile * ((mass_y - center_y) ** 2)) / mass_sum))248 249 chars = detect_char_components(tight)250 char_feats = summarize_char_geometry(chars, tight.shape[0], tight.shape[1])251 char_count = float(max(char_feats[0], 1.0))252 tall_ratio = float(char_feats[7] / char_count)253 wide_ratio = float(char_feats[8] / char_count)254 max_cy_gap_norm = float(char_feats[12] / max(tight.shape[0], 1))255 mean_char_h_norm = float(char_feats[1] / max(tight.shape[0], 1))256 257 # Pooling stretto sulla verticale: privilegia la forma delle bande e delle fusioni.258 vertical_map = cv2.resize(norm_f, (8, 24), interpolation=cv2.INTER_AREA).reshape(-1)259 overlap_feats = np.array(260 [261 upper,262 middle,263 lower,264 peak,265 valley_ratio,266 band_count,267 band_gap,268 transitions,269 float(norm_f.mean()),270 float(np.std(row_profile)),271 center_y / max(h - 1, 1),272 spread_y / max(h - 1, 1),273 float(np.max(row_grad)) if row_grad.size else 0.0,274 float(np.mean(row_grad)) if row_grad.size else 0.0,275 tall_ratio,276 wide_ratio,277 max_cy_gap_norm,278 mean_char_h_norm,279 ],280 dtype=np.float32,281 )282 283 overlap_vector = np.concatenate(284 [285 vertical_map.astype(np.float32),286 row_profile.astype(np.float32),287 row_profile_s.astype(np.float32),288 row_profile_ss.astype(np.float32),289 row_grad.astype(np.float32),290 overlap_feats.astype(np.float32),291 char_feats.astype(np.float32),292 ]293 ).astype(np.float32)294 295 # L'embedding finale e' volutamente specifica sulla sovrapposizione.296 embedding = np.concatenate(297 [298 overlap_vector,299 ]300 ).astype(np.float32)301 302 norm_value = np.linalg.norm(embedding)303 if norm_value > 0:304 embedding = embedding / norm_value305 306 debug = {307 "band_count": band_count,308 "band_gap": band_gap,309 "peak": peak,310 "valley_ratio": valley_ratio,311 "transitions": transitions,312 "char_count": float(char_feats[0]) if char_feats.size else 0.0,313 "tall_char_count": float(char_feats[7]) if char_feats.size else 0.0,314 "wide_char_count": float(char_feats[8]) if char_feats.size else 0.0,315 "max_cy_gap": float(char_feats[12]) if char_feats.size else 0.0,316 "tall_ratio": tall_ratio,317 "wide_ratio": wide_ratio,318 "max_cy_gap_norm": max_cy_gap_norm,319 "mean_char_h_norm": mean_char_h_norm,320 "center_y_norm": center_y / max(h - 1, 1),321 "spread_y_norm": spread_y / max(h - 1, 1),322 }323 return embedding, debug324 