All checks were successful
Build & Deploy / deploy (push) Successful in 1m24s
Auslesen im Hintergrund: /api/scan antwortet, sobald die Karten freigestellt und gespeichert sind, und stößt die Extraktion als Task an. Ein Zwanzigerstapel blockierte vorher den Upload für die ganze Dauer aller Modellaufrufe. Die Oberfläche zeigt "wird gelesen" und lädt nach, solange etwas offen ist. Zwei Folgen davon sind mitbehandelt: Beim Schreiben der Ergebnisse steht COALESCE, damit ein Handeintrag während des Lesens nicht überschrieben wird, und offene Karten werden beim Start nachgeholt, statt dauerhaft in der Warteschlange zu hängen. Bildaufbereitung des Zuschnitts: - Der Einzug zieht die erkannten Ecken um 1,5 % zur Mitte, damit kein Untergrund im Zuschnitt bleibt. - Die Beleuchtung wird ausgeglichen (Division durch eine weichgezeichnete Fassung), damit das Papier weiß wird statt grau. Dunkle Karten bleiben unangetastet - bei ihnen ist das Dunkle das Papier, kein Schatten. - Das Modell meldet die nötige Drehung im Schema; das gespeicherte Bild wird entsprechend gedreht. Geometrisch ist die Lage nicht bestimmbar. Sammelexport entfernt: Der Button "Alle als vCard" ist weg, mit ihm der Endpunkt /vcf sowie vcard.build_many und db.execute_many, die dadurch keinen Aufrufer mehr hatten. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
197 lines
7.4 KiB
Python
197 lines
7.4 KiB
Python
"""Visitenkarten aus einem Foto freistellen.
|
|
|
|
Reine Geometrie, kein Modell: Kanten finden, Rechtecke mit Kartenformat
|
|
behalten, perspektivisch entzerren. Der Inhalt der Karte spielt hier keine
|
|
Rolle - den liest spaeter das Vision-Modell aus dem Zuschnitt.
|
|
"""
|
|
import cv2
|
|
import numpy as np
|
|
|
|
# ISO 7810 ID-1: 85,6 x 54 mm. Toleranz nach unten fuer abweichende Formate.
|
|
CARD_RATIO = 85.6 / 54.0
|
|
RATIO_MIN, RATIO_MAX = 1.25, 2.05
|
|
|
|
OUT_WIDTH = 1400 # Obergrenze; kleinere Karten werden nicht hochskaliert
|
|
MIN_WIDTH = 600
|
|
OUT_HEIGHT = round(OUT_WIDTH / CARD_RATIO)
|
|
|
|
# Das gefundene Rechteck liegt auf der Kartenkante. Ein schmaler Streifen
|
|
# davon ist noch Tisch - der faellt weg, bevor zugeschnitten wird.
|
|
INSET = 0.015
|
|
|
|
DETECT_EDGE = 1600 # Aufloesung, auf der gesucht wird
|
|
MIN_AREA_FRACTION = 0.004 # kleiner ist Rauschen, kein Kartenfund
|
|
MAX_AREA_FRACTION = 0.60 # groesser ist der Tisch, nicht die Karte
|
|
MIN_FILL = 0.80 # Kontur muss ihr eigenes Rechteck fuellen
|
|
|
|
|
|
def _masks(gray: np.ndarray) -> list[np.ndarray]:
|
|
"""Mehrere Binaerbilder, weil je nach Untergrund ein anderes traegt."""
|
|
blurred = cv2.GaussianBlur(gray, (5, 5), 0)
|
|
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5))
|
|
|
|
edges = cv2.dilate(cv2.Canny(blurred, 40, 120), kernel, iterations=2)
|
|
edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2)
|
|
|
|
adaptive = cv2.adaptiveThreshold(
|
|
blurred, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 51, 10
|
|
)
|
|
adaptive = cv2.morphologyEx(adaptive, cv2.MORPH_CLOSE, kernel, iterations=2)
|
|
|
|
_, otsu = cv2.threshold(blurred, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
|
|
|
|
return [edges, adaptive, otsu, cv2.bitwise_not(otsu)]
|
|
|
|
|
|
def _candidates(mask: np.ndarray, image_area: float) -> list[tuple]:
|
|
contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
|
|
found = []
|
|
for contour in contours:
|
|
rect = cv2.minAreaRect(contour)
|
|
(_, _), (width, height), _ = rect
|
|
if width < 24 or height < 24:
|
|
continue
|
|
long_side, short_side = max(width, height), min(width, height)
|
|
if not RATIO_MIN <= long_side / short_side <= RATIO_MAX:
|
|
continue
|
|
rect_area = width * height
|
|
if not MIN_AREA_FRACTION * image_area <= rect_area <= MAX_AREA_FRACTION * image_area:
|
|
continue
|
|
fill = cv2.contourArea(contour) / rect_area
|
|
if fill < MIN_FILL:
|
|
continue
|
|
found.append((rect, fill))
|
|
return found
|
|
|
|
|
|
def _deduplicate(candidates: list[tuple]) -> list[tuple]:
|
|
"""Dieselbe Karte wird von mehreren Masken gefunden - besten Treffer behalten."""
|
|
kept: list[tuple] = []
|
|
for rect, fill in sorted(candidates, key=lambda c: -c[1]):
|
|
(cx, cy), (width, height), _ = rect
|
|
radius = min(width, height) * 0.5
|
|
duplicate = False
|
|
for other_rect, _ in kept:
|
|
(ox, oy), _, _ = other_rect
|
|
if np.hypot(cx - ox, cy - oy) < radius:
|
|
duplicate = True
|
|
break
|
|
if not duplicate:
|
|
kept.append((rect, fill))
|
|
return kept
|
|
|
|
|
|
def _order_quad(points: np.ndarray) -> np.ndarray:
|
|
"""Ecken als oben-links, oben-rechts, unten-rechts, unten-links."""
|
|
total = points.sum(axis=1)
|
|
diff = points[:, 1] - points[:, 0]
|
|
ordered = np.array(
|
|
[
|
|
points[np.argmin(total)], # oben links
|
|
points[np.argmin(diff)], # oben rechts
|
|
points[np.argmax(total)], # unten rechts
|
|
points[np.argmax(diff)], # unten links
|
|
],
|
|
dtype="float32",
|
|
)
|
|
# Hochkant liegende Karte um 90 Grad drehen, damit der Zuschnitt quer ist.
|
|
width = np.linalg.norm(ordered[1] - ordered[0])
|
|
height = np.linalg.norm(ordered[3] - ordered[0])
|
|
if height > width:
|
|
ordered = np.roll(ordered, -1, axis=0)
|
|
return ordered
|
|
|
|
|
|
def _reading_order(rects: list) -> list:
|
|
"""Zeilenweise sortieren, damit die Reihenfolge dem Tisch entspricht."""
|
|
if not rects:
|
|
return rects
|
|
row_height = np.median([min(r[1]) for r in rects]) * 0.7
|
|
return sorted(rects, key=lambda r: (round(r[0][1] / row_height), r[0][0]))
|
|
|
|
|
|
def segment(image: np.ndarray) -> tuple[list[np.ndarray], bool]:
|
|
"""Liefert die entzerrten Kartenbilder und ob auf das Gesamtbild
|
|
zurueckgefallen wurde (kein Kartenrechteck gefunden)."""
|
|
height, width = image.shape[:2]
|
|
scale = DETECT_EDGE / max(height, width) if max(height, width) > DETECT_EDGE else 1.0
|
|
small = (
|
|
cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA)
|
|
if scale < 1.0
|
|
else image
|
|
)
|
|
gray = cv2.cvtColor(small, cv2.COLOR_BGR2GRAY)
|
|
image_area = float(small.shape[0] * small.shape[1])
|
|
|
|
candidates: list[tuple] = []
|
|
for mask in _masks(gray):
|
|
candidates.extend(_candidates(mask, image_area))
|
|
|
|
rects = [rect for rect, _ in _deduplicate(candidates)]
|
|
if not rects:
|
|
return [_fit_whole(image)], True
|
|
|
|
crops = []
|
|
for rect in _reading_order(rects):
|
|
quad = _order_quad(cv2.boxPoints(rect)) / scale # zurueck auf volle Aufloesung
|
|
quad = _shrink(quad)
|
|
width, height = _target_size(quad)
|
|
target = np.array(
|
|
[[0, 0], [width, 0], [width, height], [0, height]], dtype="float32"
|
|
)
|
|
matrix = cv2.getPerspectiveTransform(quad, target)
|
|
crop = cv2.warpPerspective(image, matrix, (width, height), flags=cv2.INTER_CUBIC)
|
|
crops.append(flatten(crop))
|
|
return crops, False
|
|
|
|
|
|
def _shrink(quad: np.ndarray, factor: float = INSET) -> np.ndarray:
|
|
"""Ecken zur Mitte ziehen, damit kein Untergrund im Zuschnitt bleibt."""
|
|
center = quad.mean(axis=0)
|
|
return (center + (quad - center) * (1.0 - factor)).astype("float32")
|
|
|
|
|
|
def flatten(card: np.ndarray) -> np.ndarray:
|
|
"""Beleuchtung ausgleichen, damit das Papier weiss wird statt grau.
|
|
|
|
Geteilt wird durch eine stark weichgezeichnete Fassung des Bildes - das
|
|
ist die Beleuchtung. Uebrig bleibt der Aufdruck. Dunkle Karten bleiben
|
|
unangetastet: bei ihnen ist das Dunkle das Papier, kein Schatten.
|
|
"""
|
|
gray = cv2.cvtColor(card, cv2.COLOR_BGR2GRAY)
|
|
if np.median(gray) < 120:
|
|
return card
|
|
|
|
sigma = max(card.shape[1] / 16.0, 3.0)
|
|
illumination = cv2.GaussianBlur(gray, (0, 0), sigma).astype(np.float32)
|
|
gain = np.clip(250.0 / np.maximum(illumination, 1.0), 1.0, 2.2)
|
|
lifted = card.astype(np.float32) * gain[:, :, None]
|
|
return np.clip(lifted, 0, 255).astype(np.uint8)
|
|
|
|
|
|
def _target_size(quad: np.ndarray) -> tuple:
|
|
"""Zuschnitt so gross wie die Vorlage, hoechstens OUT_WIDTH.
|
|
|
|
Auf einem Stapelfoto ist eine Karte nur ein paar hundert Pixel breit.
|
|
Sie auf 1400 hochzurechnen erzeugt vier Mal so viele Pixel ohne ein
|
|
Quentchen mehr Information - und kostet Speicher, Bildgroesse und
|
|
Bildtokens beim Modellaufruf.
|
|
"""
|
|
long_edge = max(
|
|
np.linalg.norm(quad[1] - quad[0]), np.linalg.norm(quad[2] - quad[3])
|
|
)
|
|
width = int(min(OUT_WIDTH, max(MIN_WIDTH, round(long_edge))))
|
|
return width, round(width / CARD_RATIO)
|
|
|
|
|
|
def _fit_whole(image: np.ndarray) -> np.ndarray:
|
|
"""Ohne Fund: das ganze Bild als eine Karte behandeln."""
|
|
height, width = image.shape[:2]
|
|
if width < height: # Hochformat drehen
|
|
image = cv2.rotate(image, cv2.ROTATE_90_CLOCKWISE)
|
|
height, width = image.shape[:2]
|
|
scale = min(OUT_WIDTH / width, OUT_HEIGHT / height, 1.0)
|
|
if scale < 1.0:
|
|
image = cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA)
|
|
return image
|