43 lines
2.7 KiB
Python
43 lines
2.7 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""Lokale Pruefung eines Artikels vor dem Import.
|
|
Aufruf: python3 build/pruef_lokal.py artikel/artikel-SLUG.json [brief.json]"""
|
|
import json, re, subprocess, sys, os
|
|
p = sys.argv[1]
|
|
a = json.load(open(p, encoding='utf-8'))
|
|
idx = json.load(open('artikel-index.json', encoding='utf-8'))
|
|
slugs = {z['slug'] for z in idx}
|
|
bp = sys.argv[2] if len(sys.argv) > 2 else f"build/lokal/briefe/{a['slug']}.json"
|
|
c = a['content_html']
|
|
txt = re.sub(r'\[skrift_preis[^\]]*\]', '3,20 €', c)
|
|
txt = re.sub(r'\[/?ratgeber_link[^\]]*\]', '', txt)
|
|
plain = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', txt)).strip()
|
|
w = len(plain.split())
|
|
faqt = ' '.join(f['frage'] + ' ' + f['antwort'] for f in a.get('faq', []))
|
|
voll = (a.get('kurzantwort', '') + ' ' + plain + ' ' + faqt).lower()
|
|
h2 = re.findall(r'<h2>(.*?)</h2>', c); h3 = re.findall(r'<h3>(.*?)</h3>', c)
|
|
print(f"== {a['slug']} | {w} Woerter | H2 {len(h2)} | H3 {len(h3)} | Tabellen {c.count('<table')} | FAQ {len(a.get('faq',[]))}")
|
|
probleme = []
|
|
if not 900 <= w <= 1300: probleme.append(f"Laenge {w}")
|
|
if not 4 <= len(h2) <= 6: probleme.append(f"H2 {len(h2)}")
|
|
if any(x.strip()[-1:] not in '.?' for x in h2 + h3): probleme.append("Ueberschrift ohne Punkt/Fragezeichen")
|
|
if len(a['seo_titel']) > 60: probleme.append(f"SEO-Titel {len(a['seo_titel'])}")
|
|
if len(a['seo_beschreibung']) > 155: probleme.append(f"Meta {len(a['seo_beschreibung'])}")
|
|
if 'Skrift' not in a['kurzantwort']: probleme.append("Skrift fehlt in Kurzantwort")
|
|
if not a.get('faq_ueberschrift'): probleme.append("faq_ueberschrift fehlt")
|
|
if re.search('[–—]', c + a['kurzantwort'] + faqt): probleme.append("Gedankenstrich")
|
|
if re.search(r'\bich\b', plain, re.I): probleme.append("Ich-Form")
|
|
for s in re.findall(r'\[ratgeber_link slug="([^"]+)"', c):
|
|
if s not in slugs: probleme.append(f"unbekannter Slug {s}")
|
|
if a['slug'] in ('kaltakquise-per-brief', 'serienbriefe-unterschreiben', 'dsgvo-briefwerbung') and 'Keine Rechtsberatung: Dieser Beitrag gibt den Stand unserer Praxiserfahrung wieder' not in plain:
|
|
probleme.append("Disclaimer fehlt")
|
|
if os.path.exists(bp):
|
|
b = json.load(open(bp, encoding='utf-8'))
|
|
fehlt = [f"{x['begriff']}({x['min']})" for x in b['begriffe_pflicht'] if voll.count(x['begriff'].lower()) < (x['min'] or 1)]
|
|
print(f" Pflichtbegriffe: {len(b['begriffe_pflicht'])-len(fehlt)}/{len(b['begriffe_pflicht'])} fehlt: {', '.join(fehlt) or '-'}")
|
|
print(f" Zielwortzahl NW {b['wortzahl_ziel']}")
|
|
r = subprocess.run(['node', 'build/stiltest.js', p], capture_output=True, text=True)
|
|
hin = [l for l in r.stdout.splitlines()[1:] if 'Kein Praxiswissen' not in l]
|
|
for l in hin: print(' ', l.strip()[:190])
|
|
for x in probleme: print(' PROBLEM:', x)
|
|
if not hin and not probleme: print(' sauber')
|