# -*- coding: utf-8 -*-
"""Lokale Pruefung eines Artikels vor dem Import.
Aufruf: python3 build/pruef_lokal.py artikel/artikel-SLUG.json [brief.json]"""
import json, re, subprocess, sys, os
p = sys.argv[1]
a = json.load(open(p, encoding='utf-8'))
idx = json.load(open('artikel-index.json', encoding='utf-8'))
slugs = {z['slug'] for z in idx}
bp = sys.argv[2] if len(sys.argv) > 2 else f"build/lokal/briefe/{a['slug']}.json"
c = a['content_html']
txt = re.sub(r'\[skrift_preis[^\]]*\]', '3,20 €', c)
txt = re.sub(r'\[/?ratgeber_link[^\]]*\]', '', txt)
plain = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', txt)).strip()
w = len(plain.split())
faqt = ' '.join(f['frage'] + ' ' + f['antwort'] for f in a.get('faq', []))
voll = (a.get('kurzantwort', '') + ' ' + plain + ' ' + faqt).lower()
h2 = re.findall(r'
(.*?)
', c); h3 = re.findall(r'(.*?)
', c)
print(f"== {a['slug']} | {w} Woerter | H2 {len(h2)} | H3 {len(h3)} | Tabellen {c.count(' 60: probleme.append(f"SEO-Titel {len(a['seo_titel'])}")
if len(a['seo_beschreibung']) > 155: probleme.append(f"Meta {len(a['seo_beschreibung'])}")
if 'Skrift' not in a['kurzantwort']: probleme.append("Skrift fehlt in Kurzantwort")
if not a.get('faq_ueberschrift'): probleme.append("faq_ueberschrift fehlt")
if re.search('[–—]', c + a['kurzantwort'] + faqt): probleme.append("Gedankenstrich")
if re.search(r'\bich\b', plain, re.I): probleme.append("Ich-Form")
for s in re.findall(r'\[ratgeber_link slug="([^"]+)"', c):
if s not in slugs: probleme.append(f"unbekannter Slug {s}")
if a['slug'] in ('kaltakquise-per-brief',) and 'Keine Rechtsberatung: Dieser Beitrag gibt den Stand unserer Praxiserfahrung wieder' not in plain:
probleme.append("Disclaimer fehlt")
if os.path.exists(bp):
b = json.load(open(bp, encoding='utf-8'))
fehlt = [f"{x['begriff']}({x['min']})" for x in b['begriffe_pflicht'] if voll.count(x['begriff'].lower()) < (x['min'] or 1)]
print(f" Pflichtbegriffe: {len(b['begriffe_pflicht'])-len(fehlt)}/{len(b['begriffe_pflicht'])} fehlt: {', '.join(fehlt) or '-'}")
print(f" Zielwortzahl NW {b['wortzahl_ziel']}")
r = subprocess.run(['node', 'build/stiltest.js', p], capture_output=True, text=True)
hin = [l for l in r.stdout.splitlines()[1:] if 'Kein Praxiswissen' not in l]
for l in hin: print(' ', l.strip()[:190])
for x in probleme: print(' PROBLEM:', x)
if not hin and not probleme: print(' sauber')