# -*- coding: utf-8 -*- """Lokale Pruefung eines Artikels vor dem Import. Aufruf: python3 build/pruef_lokal.py artikel/artikel-SLUG.json [brief.json]""" import json, re, subprocess, sys, os p = sys.argv[1] a = json.load(open(p, encoding='utf-8')) idx = json.load(open('artikel-index.json', encoding='utf-8')) slugs = {z['slug'] for z in idx} bp = sys.argv[2] if len(sys.argv) > 2 else f"build/lokal/briefe/{a['slug']}.json" c = a['content_html'] txt = re.sub(r'\[skrift_preis[^\]]*\]', '3,20 €', c) txt = re.sub(r'\[/?ratgeber_link[^\]]*\]', '', txt) plain = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', txt)).strip() w = len(plain.split()) faqt = ' '.join(f['frage'] + ' ' + f['antwort'] for f in a.get('faq', [])) voll = (a.get('kurzantwort', '') + ' ' + plain + ' ' + faqt).lower() h2 = re.findall(r'

(.*?)

', c); h3 = re.findall(r'

(.*?)

', c) print(f"== {a['slug']} | {w} Woerter | H2 {len(h2)} | H3 {len(h3)} | Tabellen {c.count(' 60: probleme.append(f"SEO-Titel {len(a['seo_titel'])}") if len(a['seo_beschreibung']) > 155: probleme.append(f"Meta {len(a['seo_beschreibung'])}") if 'Skrift' not in a['kurzantwort']: probleme.append("Skrift fehlt in Kurzantwort") if not a.get('faq_ueberschrift'): probleme.append("faq_ueberschrift fehlt") if re.search('[–—]', c + a['kurzantwort'] + faqt): probleme.append("Gedankenstrich") if re.search(r'\bich\b', plain, re.I): probleme.append("Ich-Form") for s in re.findall(r'\[ratgeber_link slug="([^"]+)"', c): if s not in slugs: probleme.append(f"unbekannter Slug {s}") if a['slug'] in ('kaltakquise-per-brief',) and 'Keine Rechtsberatung: Dieser Beitrag gibt den Stand unserer Praxiserfahrung wieder' not in plain: probleme.append("Disclaimer fehlt") if os.path.exists(bp): b = json.load(open(bp, encoding='utf-8')) fehlt = [f"{x['begriff']}({x['min']})" for x in b['begriffe_pflicht'] if voll.count(x['begriff'].lower()) < (x['min'] or 1)] print(f" Pflichtbegriffe: {len(b['begriffe_pflicht'])-len(fehlt)}/{len(b['begriffe_pflicht'])} fehlt: {', '.join(fehlt) or '-'}") print(f" Zielwortzahl NW {b['wortzahl_ziel']}") r = subprocess.run(['node', 'build/stiltest.js', p], capture_output=True, text=True) hin = [l for l in r.stdout.splitlines()[1:] if 'Kein Praxiswissen' not in l] for l in hin: print(' ', l.strip()[:190]) for x in probleme: print(' PROBLEM:', x) if not hin and not probleme: print(' sauber')