Drei Änderungen, die den Spitzenverbrauch eines Stapelscans senken: - Zuschnitte werden nicht mehr auf 1400 px hochgerechnet. Auf einem Stapelfoto ist eine Karte nur ein paar hundert Pixel breit; sie aufzublasen erzeugt vier Mal so viele Pixel ohne mehr Information und kostet zusätzlich Bildtokens beim Modellaufruf. OUT_WIDTH ist jetzt eine Obergrenze. - Die um 180 Grad gedrehte Fassung entsteht erst, wenn ein Ergebnis leer bleibt, statt für jede Karte auf Vorrat. Das halbiert die JPEG-Kodierungen im Normalfall. - Zuschnitte werden einzeln kodiert und sofort freigegeben, statt gesammelt im Speicher zu liegen. Dazu mem_limit und cpus für den App-Container: ein Ausreißer soll den Container treffen, nicht den Host, auf dem auch Jitsi und MySQL laufen. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
169 lines
6.3 KiB
Python
169 lines
6.3 KiB
Python
"""Visitenkarten aus einem Foto freistellen.
|
|
|
|
Reine Geometrie, kein Modell: Kanten finden, Rechtecke mit Kartenformat
|
|
behalten, perspektivisch entzerren. Der Inhalt der Karte spielt hier keine
|
|
Rolle - den liest spaeter das Vision-Modell aus dem Zuschnitt.
|
|
"""
|
|
import cv2
|
|
import numpy as np
|
|
|
|
# ISO 7810 ID-1: 85,6 x 54 mm. Toleranz nach unten fuer abweichende Formate.
|
|
CARD_RATIO = 85.6 / 54.0
|
|
RATIO_MIN, RATIO_MAX = 1.25, 2.05
|
|
|
|
OUT_WIDTH = 1400 # Obergrenze; kleinere Karten werden nicht hochskaliert
|
|
MIN_WIDTH = 600
|
|
OUT_HEIGHT = round(OUT_WIDTH / CARD_RATIO)
|
|
|
|
DETECT_EDGE = 1600 # Aufloesung, auf der gesucht wird
|
|
MIN_AREA_FRACTION = 0.004 # kleiner ist Rauschen, kein Kartenfund
|
|
MAX_AREA_FRACTION = 0.60 # groesser ist der Tisch, nicht die Karte
|
|
MIN_FILL = 0.80 # Kontur muss ihr eigenes Rechteck fuellen
|
|
|
|
|
|
def _masks(gray: np.ndarray) -> list[np.ndarray]:
|
|
"""Mehrere Binaerbilder, weil je nach Untergrund ein anderes traegt."""
|
|
blurred = cv2.GaussianBlur(gray, (5, 5), 0)
|
|
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5))
|
|
|
|
edges = cv2.dilate(cv2.Canny(blurred, 40, 120), kernel, iterations=2)
|
|
edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2)
|
|
|
|
adaptive = cv2.adaptiveThreshold(
|
|
blurred, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 51, 10
|
|
)
|
|
adaptive = cv2.morphologyEx(adaptive, cv2.MORPH_CLOSE, kernel, iterations=2)
|
|
|
|
_, otsu = cv2.threshold(blurred, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
|
|
|
|
return [edges, adaptive, otsu, cv2.bitwise_not(otsu)]
|
|
|
|
|
|
def _candidates(mask: np.ndarray, image_area: float) -> list[tuple]:
|
|
contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
|
|
found = []
|
|
for contour in contours:
|
|
rect = cv2.minAreaRect(contour)
|
|
(_, _), (width, height), _ = rect
|
|
if width < 24 or height < 24:
|
|
continue
|
|
long_side, short_side = max(width, height), min(width, height)
|
|
if not RATIO_MIN <= long_side / short_side <= RATIO_MAX:
|
|
continue
|
|
rect_area = width * height
|
|
if not MIN_AREA_FRACTION * image_area <= rect_area <= MAX_AREA_FRACTION * image_area:
|
|
continue
|
|
fill = cv2.contourArea(contour) / rect_area
|
|
if fill < MIN_FILL:
|
|
continue
|
|
found.append((rect, fill))
|
|
return found
|
|
|
|
|
|
def _deduplicate(candidates: list[tuple]) -> list[tuple]:
|
|
"""Dieselbe Karte wird von mehreren Masken gefunden - besten Treffer behalten."""
|
|
kept: list[tuple] = []
|
|
for rect, fill in sorted(candidates, key=lambda c: -c[1]):
|
|
(cx, cy), (width, height), _ = rect
|
|
radius = min(width, height) * 0.5
|
|
duplicate = False
|
|
for other_rect, _ in kept:
|
|
(ox, oy), _, _ = other_rect
|
|
if np.hypot(cx - ox, cy - oy) < radius:
|
|
duplicate = True
|
|
break
|
|
if not duplicate:
|
|
kept.append((rect, fill))
|
|
return kept
|
|
|
|
|
|
def _order_quad(points: np.ndarray) -> np.ndarray:
|
|
"""Ecken als oben-links, oben-rechts, unten-rechts, unten-links."""
|
|
total = points.sum(axis=1)
|
|
diff = points[:, 1] - points[:, 0]
|
|
ordered = np.array(
|
|
[
|
|
points[np.argmin(total)], # oben links
|
|
points[np.argmin(diff)], # oben rechts
|
|
points[np.argmax(total)], # unten rechts
|
|
points[np.argmax(diff)], # unten links
|
|
],
|
|
dtype="float32",
|
|
)
|
|
# Hochkant liegende Karte um 90 Grad drehen, damit der Zuschnitt quer ist.
|
|
width = np.linalg.norm(ordered[1] - ordered[0])
|
|
height = np.linalg.norm(ordered[3] - ordered[0])
|
|
if height > width:
|
|
ordered = np.roll(ordered, -1, axis=0)
|
|
return ordered
|
|
|
|
|
|
def _reading_order(rects: list) -> list:
|
|
"""Zeilenweise sortieren, damit die Reihenfolge dem Tisch entspricht."""
|
|
if not rects:
|
|
return rects
|
|
row_height = np.median([min(r[1]) for r in rects]) * 0.7
|
|
return sorted(rects, key=lambda r: (round(r[0][1] / row_height), r[0][0]))
|
|
|
|
|
|
def segment(image: np.ndarray) -> tuple[list[np.ndarray], bool]:
|
|
"""Liefert die entzerrten Kartenbilder und ob auf das Gesamtbild
|
|
zurueckgefallen wurde (kein Kartenrechteck gefunden)."""
|
|
height, width = image.shape[:2]
|
|
scale = DETECT_EDGE / max(height, width) if max(height, width) > DETECT_EDGE else 1.0
|
|
small = (
|
|
cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA)
|
|
if scale < 1.0
|
|
else image
|
|
)
|
|
gray = cv2.cvtColor(small, cv2.COLOR_BGR2GRAY)
|
|
image_area = float(small.shape[0] * small.shape[1])
|
|
|
|
candidates: list[tuple] = []
|
|
for mask in _masks(gray):
|
|
candidates.extend(_candidates(mask, image_area))
|
|
|
|
rects = [rect for rect, _ in _deduplicate(candidates)]
|
|
if not rects:
|
|
return [_fit_whole(image)], True
|
|
|
|
crops = []
|
|
for rect in _reading_order(rects):
|
|
quad = _order_quad(cv2.boxPoints(rect)) / scale # zurueck auf volle Aufloesung
|
|
width, height = _target_size(quad)
|
|
target = np.array(
|
|
[[0, 0], [width, 0], [width, height], [0, height]], dtype="float32"
|
|
)
|
|
matrix = cv2.getPerspectiveTransform(quad, target)
|
|
crops.append(
|
|
cv2.warpPerspective(image, matrix, (width, height), flags=cv2.INTER_CUBIC)
|
|
)
|
|
return crops, False
|
|
|
|
|
|
def _target_size(quad: np.ndarray) -> tuple:
|
|
"""Zuschnitt so gross wie die Vorlage, hoechstens OUT_WIDTH.
|
|
|
|
Auf einem Stapelfoto ist eine Karte nur ein paar hundert Pixel breit.
|
|
Sie auf 1400 hochzurechnen erzeugt vier Mal so viele Pixel ohne ein
|
|
Quentchen mehr Information - und kostet Speicher, Bildgroesse und
|
|
Bildtokens beim Modellaufruf.
|
|
"""
|
|
long_edge = max(
|
|
np.linalg.norm(quad[1] - quad[0]), np.linalg.norm(quad[2] - quad[3])
|
|
)
|
|
width = int(min(OUT_WIDTH, max(MIN_WIDTH, round(long_edge))))
|
|
return width, round(width / CARD_RATIO)
|
|
|
|
|
|
def _fit_whole(image: np.ndarray) -> np.ndarray:
|
|
"""Ohne Fund: das ganze Bild als eine Karte behandeln."""
|
|
height, width = image.shape[:2]
|
|
if width < height: # Hochformat drehen
|
|
image = cv2.rotate(image, cv2.ROTATE_90_CLOCKWISE)
|
|
height, width = image.shape[:2]
|
|
scale = min(OUT_WIDTH / width, OUT_HEIGHT / height, 1.0)
|
|
if scale < 1.0:
|
|
image = cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA)
|
|
return image
|