Files
Skrift-Auto-Article/build/pruef_lokal.py

43 lines
2.7 KiB
Python

# -*- coding: utf-8 -*-
"""Lokale Pruefung eines Artikels vor dem Import.
Aufruf: python3 build/pruef_lokal.py artikel/artikel-SLUG.json [brief.json]"""
import json, re, subprocess, sys, os
p = sys.argv[1]
a = json.load(open(p, encoding='utf-8'))
idx = json.load(open('artikel-index.json', encoding='utf-8'))
slugs = {z['slug'] for z in idx}
bp = sys.argv[2] if len(sys.argv) > 2 else f"build/lokal/briefe/{a['slug']}.json"
c = a['content_html']
txt = re.sub(r'\[skrift_preis[^\]]*\]', '3,20 €', c)
txt = re.sub(r'\[/?ratgeber_link[^\]]*\]', '', txt)
plain = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', txt)).strip()
w = len(plain.split())
faqt = ' '.join(f['frage'] + ' ' + f['antwort'] for f in a.get('faq', []))
voll = (a.get('kurzantwort', '') + ' ' + plain + ' ' + faqt).lower()
h2 = re.findall(r'<h2>(.*?)</h2>', c); h3 = re.findall(r'<h3>(.*?)</h3>', c)
print(f"== {a['slug']} | {w} Woerter | H2 {len(h2)} | H3 {len(h3)} | Tabellen {c.count('<table')} | FAQ {len(a.get('faq',[]))}")
probleme = []
if not 900 <= w <= 1300: probleme.append(f"Laenge {w}")
if not 4 <= len(h2) <= 6: probleme.append(f"H2 {len(h2)}")
if any(x.strip()[-1:] not in '.?' for x in h2 + h3): probleme.append("Ueberschrift ohne Punkt/Fragezeichen")
if len(a['seo_titel']) > 60: probleme.append(f"SEO-Titel {len(a['seo_titel'])}")
if len(a['seo_beschreibung']) > 155: probleme.append(f"Meta {len(a['seo_beschreibung'])}")
if 'Skrift' not in a['kurzantwort']: probleme.append("Skrift fehlt in Kurzantwort")
if not a.get('faq_ueberschrift'): probleme.append("faq_ueberschrift fehlt")
if re.search('[–—]', c + a['kurzantwort'] + faqt): probleme.append("Gedankenstrich")
if re.search(r'\bich\b', plain, re.I): probleme.append("Ich-Form")
for s in re.findall(r'\[ratgeber_link slug="([^"]+)"', c):
if s not in slugs: probleme.append(f"unbekannter Slug {s}")
if a['slug'] in ('kaltakquise-per-brief', 'serienbriefe-unterschreiben', 'dsgvo-briefwerbung') and 'Keine Rechtsberatung: Dieser Beitrag gibt den Stand unserer Praxiserfahrung wieder' not in plain:
probleme.append("Disclaimer fehlt")
if os.path.exists(bp):
b = json.load(open(bp, encoding='utf-8'))
fehlt = [f"{x['begriff']}({x['min']})" for x in b['begriffe_pflicht'] if voll.count(x['begriff'].lower()) < (x['min'] or 1)]
print(f" Pflichtbegriffe: {len(b['begriffe_pflicht'])-len(fehlt)}/{len(b['begriffe_pflicht'])} fehlt: {', '.join(fehlt) or '-'}")
print(f" Zielwortzahl NW {b['wortzahl_ziel']}")
r = subprocess.run(['node', 'build/stiltest.js', p], capture_output=True, text=True)
hin = [l for l in r.stdout.splitlines()[1:] if 'Kein Praxiswissen' not in l]
for l in hin: print(' ', l.strip()[:190])
for x in probleme: print(' PROBLEM:', x)
if not hin and not probleme: print(' sauber')