"""Visitenkarten aus einem Foto freistellen. Reine Geometrie, kein Modell: Kanten finden, Rechtecke mit Kartenformat behalten, perspektivisch entzerren. Der Inhalt der Karte spielt hier keine Rolle - den liest spaeter das Vision-Modell aus dem Zuschnitt. """ import cv2 import numpy as np # ISO 7810 ID-1: 85,6 x 54 mm. Toleranz nach unten fuer abweichende Formate. CARD_RATIO = 85.6 / 54.0 RATIO_MIN, RATIO_MAX = 1.25, 2.05 OUT_WIDTH = 1400 # Obergrenze; kleinere Karten werden nicht hochskaliert MIN_WIDTH = 600 OUT_HEIGHT = round(OUT_WIDTH / CARD_RATIO) # Das gefundene Rechteck liegt auf der Kartenkante. Ein schmaler Streifen # davon ist noch Tisch - der faellt weg, bevor zugeschnitten wird. INSET = 0.015 DETECT_EDGE = 1600 # Aufloesung, auf der gesucht wird MIN_AREA_FRACTION = 0.004 # kleiner ist Rauschen, kein Kartenfund MAX_AREA_FRACTION = 0.60 # groesser ist der Tisch, nicht die Karte MIN_FILL = 0.80 # Kontur muss ihr eigenes Rechteck fuellen def _masks(gray: np.ndarray) -> list[np.ndarray]: """Mehrere Binaerbilder, weil je nach Untergrund ein anderes traegt.""" blurred = cv2.GaussianBlur(gray, (5, 5), 0) kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5)) edges = cv2.dilate(cv2.Canny(blurred, 40, 120), kernel, iterations=2) edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2) adaptive = cv2.adaptiveThreshold( blurred, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 51, 10 ) adaptive = cv2.morphologyEx(adaptive, cv2.MORPH_CLOSE, kernel, iterations=2) _, otsu = cv2.threshold(blurred, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) return [edges, adaptive, otsu, cv2.bitwise_not(otsu)] def _candidates(mask: np.ndarray, image_area: float) -> list[tuple]: contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) found = [] for contour in contours: rect = cv2.minAreaRect(contour) (_, _), (width, height), _ = rect if width < 24 or height < 24: continue long_side, short_side = max(width, height), min(width, height) if not RATIO_MIN <= long_side / short_side <= RATIO_MAX: continue rect_area = width * height if not MIN_AREA_FRACTION * image_area <= rect_area <= MAX_AREA_FRACTION * image_area: continue fill = cv2.contourArea(contour) / rect_area if fill < MIN_FILL: continue found.append((rect, fill)) return found def _deduplicate(candidates: list[tuple]) -> list[tuple]: """Dieselbe Karte wird von mehreren Masken gefunden - besten Treffer behalten.""" kept: list[tuple] = [] for rect, fill in sorted(candidates, key=lambda c: -c[1]): (cx, cy), (width, height), _ = rect radius = min(width, height) * 0.5 duplicate = False for other_rect, _ in kept: (ox, oy), _, _ = other_rect if np.hypot(cx - ox, cy - oy) < radius: duplicate = True break if not duplicate: kept.append((rect, fill)) return kept def _order_quad(points: np.ndarray) -> np.ndarray: """Ecken als oben-links, oben-rechts, unten-rechts, unten-links.""" total = points.sum(axis=1) diff = points[:, 1] - points[:, 0] ordered = np.array( [ points[np.argmin(total)], # oben links points[np.argmin(diff)], # oben rechts points[np.argmax(total)], # unten rechts points[np.argmax(diff)], # unten links ], dtype="float32", ) # Hochkant liegende Karte um 90 Grad drehen, damit der Zuschnitt quer ist. width = np.linalg.norm(ordered[1] - ordered[0]) height = np.linalg.norm(ordered[3] - ordered[0]) if height > width: ordered = np.roll(ordered, -1, axis=0) return ordered def _reading_order(rects: list) -> list: """Zeilenweise sortieren, damit die Reihenfolge dem Tisch entspricht.""" if not rects: return rects row_height = np.median([min(r[1]) for r in rects]) * 0.7 return sorted(rects, key=lambda r: (round(r[0][1] / row_height), r[0][0])) def segment(image: np.ndarray) -> tuple[list[np.ndarray], bool]: """Liefert die entzerrten Kartenbilder und ob auf das Gesamtbild zurueckgefallen wurde (kein Kartenrechteck gefunden).""" height, width = image.shape[:2] scale = DETECT_EDGE / max(height, width) if max(height, width) > DETECT_EDGE else 1.0 small = ( cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA) if scale < 1.0 else image ) gray = cv2.cvtColor(small, cv2.COLOR_BGR2GRAY) image_area = float(small.shape[0] * small.shape[1]) candidates: list[tuple] = [] for mask in _masks(gray): candidates.extend(_candidates(mask, image_area)) rects = [rect for rect, _ in _deduplicate(candidates)] if not rects: return [_fit_whole(image)], True crops = [] for rect in _reading_order(rects): quad = _order_quad(cv2.boxPoints(rect)) / scale # zurueck auf volle Aufloesung quad = _shrink(quad) width, height = _target_size(quad) target = np.array( [[0, 0], [width, 0], [width, height], [0, height]], dtype="float32" ) matrix = cv2.getPerspectiveTransform(quad, target) crop = cv2.warpPerspective(image, matrix, (width, height), flags=cv2.INTER_CUBIC) crops.append(flatten(crop)) return crops, False def _shrink(quad: np.ndarray, factor: float = INSET) -> np.ndarray: """Ecken zur Mitte ziehen, damit kein Untergrund im Zuschnitt bleibt.""" center = quad.mean(axis=0) return (center + (quad - center) * (1.0 - factor)).astype("float32") def flatten(card: np.ndarray) -> np.ndarray: """Beleuchtung ausgleichen, damit das Papier weiss wird statt grau. Geteilt wird durch eine stark weichgezeichnete Fassung des Bildes - das ist die Beleuchtung. Uebrig bleibt der Aufdruck. Dunkle Karten bleiben unangetastet: bei ihnen ist das Dunkle das Papier, kein Schatten. """ gray = cv2.cvtColor(card, cv2.COLOR_BGR2GRAY) if np.median(gray) < 120: return card sigma = max(card.shape[1] / 16.0, 3.0) illumination = cv2.GaussianBlur(gray, (0, 0), sigma).astype(np.float32) gain = np.clip(250.0 / np.maximum(illumination, 1.0), 1.0, 2.2) lifted = card.astype(np.float32) * gain[:, :, None] return np.clip(lifted, 0, 255).astype(np.uint8) def _target_size(quad: np.ndarray) -> tuple: """Zuschnitt so gross wie die Vorlage, hoechstens OUT_WIDTH. Auf einem Stapelfoto ist eine Karte nur ein paar hundert Pixel breit. Sie auf 1400 hochzurechnen erzeugt vier Mal so viele Pixel ohne ein Quentchen mehr Information - und kostet Speicher, Bildgroesse und Bildtokens beim Modellaufruf. """ long_edge = max( np.linalg.norm(quad[1] - quad[0]), np.linalg.norm(quad[2] - quad[3]) ) width = int(min(OUT_WIDTH, max(MIN_WIDTH, round(long_edge)))) return width, round(width / CARD_RATIO) def _fit_whole(image: np.ndarray) -> np.ndarray: """Ohne Fund: das ganze Bild als eine Karte behandeln.""" height, width = image.shape[:2] if width < height: # Hochformat drehen image = cv2.rotate(image, cv2.ROTATE_90_CLOCKWISE) height, width = image.shape[:2] scale = min(OUT_WIDTH / width, OUT_HEIGHT / height, 1.0) if scale < 1.0: image = cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA) return image