Files
business-card-scanner/app/segment.py
Lucas Orth ab993e98e1 Visitenkarten-Scanner: Stapelscan, Extraktion, Übersicht, vCard-Export
Selbst gehostete PWA, die Visitenkarten von einem Foto freistellt, ausliest
und als Kontakt bereitstellt.

- Stapelscan: OpenCV findet die Kartenrechtecke über mehrere Binärmasken,
  entzerrt sie perspektivisch und schneidet sie einzeln aus. Ohne Fund gilt
  das ganze Foto als eine Karte.
- Extraktion: ein Aufruf je Zuschnitt an das Vision-Modell mit
  JSON-Schema. Kein vorgeschaltetes OCR - das würde Layout und
  Schriftgrößen wegwerfen, aus denen die Feldzuordnung entsteht.
- Metadaten: Aufnahmezeit und GPS aus den EXIF-Daten des Fotos, Ortsname
  über Nominatim, Browserstandort nur als Rückfallebene.
- Übersicht mit Volltextsuche und Filtern, Detailansicht mit Korrekturmaske.
- Notizfeld je Karte, Erinnerungen per Mail inklusive Nachholen verpasster
  Termine nach einem Neustart.
- vCard 3.0 einzeln und als Sammeldatei, Karte gilt danach als exportiert.
- Anmeldung über ein Passwort, Sitzung als signiertes Cookie.
- Deployment per Tag-Push nach den Konventionen in DEPLOY.md.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 11:45:56 +02:00

155 lines
5.7 KiB
Python

"""Visitenkarten aus einem Foto freistellen.
Reine Geometrie, kein Modell: Kanten finden, Rechtecke mit Kartenformat
behalten, perspektivisch entzerren. Der Inhalt der Karte spielt hier keine
Rolle - den liest spaeter das Vision-Modell aus dem Zuschnitt.
"""
import cv2
import numpy as np
# ISO 7810 ID-1: 85,6 x 54 mm. Toleranz nach unten fuer abweichende Formate.
CARD_RATIO = 85.6 / 54.0
RATIO_MIN, RATIO_MAX = 1.25, 2.05
OUT_WIDTH = 1400
OUT_HEIGHT = round(OUT_WIDTH / CARD_RATIO)
DETECT_EDGE = 1600 # Aufloesung, auf der gesucht wird
MIN_AREA_FRACTION = 0.004 # kleiner ist Rauschen, kein Kartenfund
MAX_AREA_FRACTION = 0.60 # groesser ist der Tisch, nicht die Karte
MIN_FILL = 0.80 # Kontur muss ihr eigenes Rechteck fuellen
def _masks(gray: np.ndarray) -> list[np.ndarray]:
"""Mehrere Binaerbilder, weil je nach Untergrund ein anderes traegt."""
blurred = cv2.GaussianBlur(gray, (5, 5), 0)
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5))
edges = cv2.dilate(cv2.Canny(blurred, 40, 120), kernel, iterations=2)
edges = cv2.morphologyEx(edges, cv2.MORPH_CLOSE, kernel, iterations=2)
adaptive = cv2.adaptiveThreshold(
blurred, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 51, 10
)
adaptive = cv2.morphologyEx(adaptive, cv2.MORPH_CLOSE, kernel, iterations=2)
_, otsu = cv2.threshold(blurred, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
return [edges, adaptive, otsu, cv2.bitwise_not(otsu)]
def _candidates(mask: np.ndarray, image_area: float) -> list[tuple]:
contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
found = []
for contour in contours:
rect = cv2.minAreaRect(contour)
(_, _), (width, height), _ = rect
if width < 24 or height < 24:
continue
long_side, short_side = max(width, height), min(width, height)
if not RATIO_MIN <= long_side / short_side <= RATIO_MAX:
continue
rect_area = width * height
if not MIN_AREA_FRACTION * image_area <= rect_area <= MAX_AREA_FRACTION * image_area:
continue
fill = cv2.contourArea(contour) / rect_area
if fill < MIN_FILL:
continue
found.append((rect, fill))
return found
def _deduplicate(candidates: list[tuple]) -> list[tuple]:
"""Dieselbe Karte wird von mehreren Masken gefunden - besten Treffer behalten."""
kept: list[tuple] = []
for rect, fill in sorted(candidates, key=lambda c: -c[1]):
(cx, cy), (width, height), _ = rect
radius = min(width, height) * 0.5
duplicate = False
for other_rect, _ in kept:
(ox, oy), _, _ = other_rect
if np.hypot(cx - ox, cy - oy) < radius:
duplicate = True
break
if not duplicate:
kept.append((rect, fill))
return kept
def _order_quad(points: np.ndarray) -> np.ndarray:
"""Ecken als oben-links, oben-rechts, unten-rechts, unten-links."""
total = points.sum(axis=1)
diff = points[:, 1] - points[:, 0]
ordered = np.array(
[
points[np.argmin(total)], # oben links
points[np.argmin(diff)], # oben rechts
points[np.argmax(total)], # unten rechts
points[np.argmax(diff)], # unten links
],
dtype="float32",
)
# Hochkant liegende Karte um 90 Grad drehen, damit der Zuschnitt quer ist.
width = np.linalg.norm(ordered[1] - ordered[0])
height = np.linalg.norm(ordered[3] - ordered[0])
if height > width:
ordered = np.roll(ordered, -1, axis=0)
return ordered
def _reading_order(rects: list) -> list:
"""Zeilenweise sortieren, damit die Reihenfolge dem Tisch entspricht."""
if not rects:
return rects
row_height = np.median([min(r[1]) for r in rects]) * 0.7
return sorted(rects, key=lambda r: (round(r[0][1] / row_height), r[0][0]))
def segment(image: np.ndarray) -> tuple[list[np.ndarray], bool]:
"""Liefert die entzerrten Kartenbilder und ob auf das Gesamtbild
zurueckgefallen wurde (kein Kartenrechteck gefunden)."""
height, width = image.shape[:2]
scale = DETECT_EDGE / max(height, width) if max(height, width) > DETECT_EDGE else 1.0
small = (
cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA)
if scale < 1.0
else image
)
gray = cv2.cvtColor(small, cv2.COLOR_BGR2GRAY)
image_area = float(small.shape[0] * small.shape[1])
candidates: list[tuple] = []
for mask in _masks(gray):
candidates.extend(_candidates(mask, image_area))
rects = [rect for rect, _ in _deduplicate(candidates)]
if not rects:
return [_fit_whole(image)], True
crops = []
for rect in _reading_order(rects):
quad = _order_quad(cv2.boxPoints(rect)) / scale # zurueck auf volle Aufloesung
target = np.array(
[[0, 0], [OUT_WIDTH, 0], [OUT_WIDTH, OUT_HEIGHT], [0, OUT_HEIGHT]],
dtype="float32",
)
matrix = cv2.getPerspectiveTransform(quad, target)
crops.append(
cv2.warpPerspective(
image, matrix, (OUT_WIDTH, OUT_HEIGHT), flags=cv2.INTER_CUBIC
)
)
return crops, False
def _fit_whole(image: np.ndarray) -> np.ndarray:
"""Ohne Fund: das ganze Bild als eine Karte behandeln."""
height, width = image.shape[:2]
if width < height: # Hochformat drehen
image = cv2.rotate(image, cv2.ROTATE_90_CLOCKWISE)
height, width = image.shape[:2]
scale = min(OUT_WIDTH / width, OUT_HEIGHT / height, 1.0)
if scale < 1.0:
image = cv2.resize(image, None, fx=scale, fy=scale, interpolation=cv2.INTER_AREA)
return image