Artikel 1-10, Stilpruefung, Querverweise, Upload-Skript
This commit is contained in:
17
build/artikel_hochladen.py
Normal file
17
build/artikel_hochladen.py
Normal file
@@ -0,0 +1,17 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Laedt alle lokalen artikel/artikel-*.json per Contents-API nach Gitea (neu oder mit sha)."""
|
||||
import base64, glob, json, os, sys
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
exec(open(os.path.join(os.path.dirname(__file__), 'push_gitea.py'), encoding='utf-8').read().split('def sammle')[0])
|
||||
|
||||
for p in sorted(glob.glob('artikel/artikel-*.json')):
|
||||
json.load(open(p, encoding='utf-8')) # nur gueltiges JSON hochladen
|
||||
inhalt = base64.b64encode(open(p, 'rb').read()).decode()
|
||||
code, d = ruf('GET', f"{REPO}/contents/{p}?ref=main")
|
||||
daten = {'content': inhalt, 'message': f'Artikel {os.path.basename(p)}', 'branch': 'main'}
|
||||
if code == 200:
|
||||
daten['sha'] = d['sha']
|
||||
code, d = ruf('PUT', f"{REPO}/contents/{p}", daten)
|
||||
else:
|
||||
code, d = ruf('POST', f"{REPO}/contents/{p}", daten)
|
||||
print(code, p, '' if code in (200, 201) else str(d)[:200])
|
||||
@@ -211,9 +211,14 @@ FERTIG = iff("Analyse fertig?", [1780, 420],
|
||||
link(ABFRAGE, FERTIG, out=0)
|
||||
|
||||
ZAEHLEN = code("Versuch zaehlen", [1780, 660], """
|
||||
// Zaehler je Query in den Workflow-Daten. $runIndex zaehlt ueber alle Artikel
|
||||
// eines Laufs hinweg und taugt deshalb nicht als Zaehler pro Artikel.
|
||||
const poll = $('Poll vorbereiten').first().json;
|
||||
const antwort = $input.first().json;
|
||||
const versuch = (poll.versuch || 0) + $runIndex + 1;
|
||||
const sd = $getWorkflowStaticData('global');
|
||||
sd.poll = sd.poll || {};
|
||||
sd.poll[poll.query] = (sd.poll[poll.query] || 0) + 1;
|
||||
const versuch = sd.poll[poll.query];
|
||||
return [{ json: { ...poll, versuch, letzter_status: antwort.status || 'unbekannt' } }];
|
||||
""")
|
||||
link(FERTIG, ZAEHLEN, out=1)
|
||||
@@ -225,6 +230,7 @@ link(ZAEHLEN, LIMIT)
|
||||
|
||||
TIMEOUT = code("Timeout melden", [2220, 780], """
|
||||
const a = $input.first().json;
|
||||
{ const sd = $getWorkflowStaticData('global'); if (sd.poll) delete sd.poll[a.query]; }
|
||||
return [{ json: { slug: a.slug, keyword: a.keyword, query: a.query, ergebnis: 'fehler',
|
||||
stufe: 'poll-timeout',
|
||||
meldung: 'NeuronWriter-Analyse fuer "' + a.keyword + '" war nach ' + a.versuch
|
||||
@@ -242,6 +248,7 @@ BRIEF = code("Brief bauen", [2220, 300], r"""
|
||||
// Feldnamen stammen aus einer echten /get-query-Antwort, siehe docs/feldnachweis.md
|
||||
const a = $('Poll vorbereiten').first().json;
|
||||
const n = $input.first().json;
|
||||
{ const sd = $getWorkflowStaticData('global'); if (sd.poll) delete sd.poll[a.query]; }
|
||||
|
||||
const liste = (arr) => (arr || []).map(x => ({
|
||||
begriff: x.t,
|
||||
@@ -318,6 +325,10 @@ const brief = {
|
||||
+ 'artikel.cta_ziel mindestens einmal im Fliesstext verlinken, mindestens zwei Business-Links insgesamt. '
|
||||
+ 'papier-schriftmuster nur ergaenzend. Keine Links auf Privatkunden-Seiten, review, agb, datenschutz, '
|
||||
+ 'impressum, zahlungsmethoden, lieferung-versand.',
|
||||
querverweise: 'Mindestens zwei, hoechstens vier Querverweise auf verwandte Ratgeberartikel, immer als '
|
||||
+ '[ratgeber_link slug="SLUG"]Ankertext[/ratgeber_link], nie als harter Link. Slugs aus artikel-index.json. '
|
||||
+ 'Der Shortcode verlinkt erst, wenn der Zielartikel veroeffentlicht ist. Ankertext beschreibt das Ziel, '
|
||||
+ 'nie "hier" oder "mehr dazu".',
|
||||
bausteine: 'Kein Block "Auch fuer Privatpersonen." und kein Abschnitt "Kostenloses Handschriftmuster '
|
||||
+ 'anfordern." im Artikeltext, beides liefert das Seitenlayout. Der Artikel endet mit dem letzten '
|
||||
+ 'inhaltlichen Abschnitt, nie mit einer Zusammenfassung.',
|
||||
|
||||
42
build/pruef_lokal.py
Normal file
42
build/pruef_lokal.py
Normal file
@@ -0,0 +1,42 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Lokale Pruefung eines Artikels vor dem Import.
|
||||
Aufruf: python3 build/pruef_lokal.py artikel/artikel-SLUG.json [brief.json]"""
|
||||
import json, re, subprocess, sys, os
|
||||
p = sys.argv[1]
|
||||
a = json.load(open(p, encoding='utf-8'))
|
||||
idx = json.load(open('artikel-index.json', encoding='utf-8'))
|
||||
slugs = {z['slug'] for z in idx}
|
||||
bp = sys.argv[2] if len(sys.argv) > 2 else f"build/lokal/briefe/{a['slug']}.json"
|
||||
c = a['content_html']
|
||||
txt = re.sub(r'\[skrift_preis[^\]]*\]', '3,20 €', c)
|
||||
txt = re.sub(r'\[/?ratgeber_link[^\]]*\]', '', txt)
|
||||
plain = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', txt)).strip()
|
||||
w = len(plain.split())
|
||||
faqt = ' '.join(f['frage'] + ' ' + f['antwort'] for f in a.get('faq', []))
|
||||
voll = (a.get('kurzantwort', '') + ' ' + plain + ' ' + faqt).lower()
|
||||
h2 = re.findall(r'<h2>(.*?)</h2>', c); h3 = re.findall(r'<h3>(.*?)</h3>', c)
|
||||
print(f"== {a['slug']} | {w} Woerter | H2 {len(h2)} | H3 {len(h3)} | Tabellen {c.count('<table')} | FAQ {len(a.get('faq',[]))}")
|
||||
probleme = []
|
||||
if not 900 <= w <= 1300: probleme.append(f"Laenge {w}")
|
||||
if not 4 <= len(h2) <= 6: probleme.append(f"H2 {len(h2)}")
|
||||
if any(x.strip()[-1:] not in '.?' for x in h2 + h3): probleme.append("Ueberschrift ohne Punkt/Fragezeichen")
|
||||
if len(a['seo_titel']) > 60: probleme.append(f"SEO-Titel {len(a['seo_titel'])}")
|
||||
if len(a['seo_beschreibung']) > 155: probleme.append(f"Meta {len(a['seo_beschreibung'])}")
|
||||
if 'Skrift' not in a['kurzantwort']: probleme.append("Skrift fehlt in Kurzantwort")
|
||||
if not a.get('faq_ueberschrift'): probleme.append("faq_ueberschrift fehlt")
|
||||
if re.search('[–—]', c + a['kurzantwort'] + faqt): probleme.append("Gedankenstrich")
|
||||
if re.search(r'\bich\b', plain, re.I): probleme.append("Ich-Form")
|
||||
for s in re.findall(r'\[ratgeber_link slug="([^"]+)"', c):
|
||||
if s not in slugs: probleme.append(f"unbekannter Slug {s}")
|
||||
if a['slug'] in ('kaltakquise-per-brief',) and 'Keine Rechtsberatung: Dieser Beitrag gibt den Stand unserer Praxiserfahrung wieder' not in plain:
|
||||
probleme.append("Disclaimer fehlt")
|
||||
if os.path.exists(bp):
|
||||
b = json.load(open(bp, encoding='utf-8'))
|
||||
fehlt = [f"{x['begriff']}({x['min']})" for x in b['begriffe_pflicht'] if voll.count(x['begriff'].lower()) < (x['min'] or 1)]
|
||||
print(f" Pflichtbegriffe: {len(b['begriffe_pflicht'])-len(fehlt)}/{len(b['begriffe_pflicht'])} fehlt: {', '.join(fehlt) or '-'}")
|
||||
print(f" Zielwortzahl NW {b['wortzahl_ziel']}")
|
||||
r = subprocess.run(['node', 'build/stiltest.js', p], capture_output=True, text=True)
|
||||
hin = [l for l in r.stdout.splitlines()[1:] if 'Kein Praxiswissen' not in l]
|
||||
for l in hin: print(' ', l.strip()[:190])
|
||||
for x in probleme: print(' PROBLEM:', x)
|
||||
if not hin and not probleme: print(' sauber')
|
||||
@@ -10,7 +10,7 @@ import base64, json, os, sys, urllib.request, urllib.error
|
||||
|
||||
# Von den Workflows gepflegt, nie von hier ueberschreiben:
|
||||
NICHT_UEBERSCHREIBEN = {'artikel-index.json', 'status-importe.json'}
|
||||
NICHT_UEBERSCHREIBEN_ORDNER = ('briefe/', 'nachbessern/', 'artikel/')
|
||||
NICHT_UEBERSCHREIBEN_ORDNER = ('briefe/', 'nachbessern/', 'artikel/', 'build/lokal/')
|
||||
AUS_ORDNER = {'.git', 'node_modules', '__pycache__'}
|
||||
AUS_DATEIEN = {'.env'}
|
||||
|
||||
|
||||
@@ -116,6 +116,14 @@ function stilpruefung(artikel) {
|
||||
// Bausteine, die das Seitenlayout bereits liefert
|
||||
if (/Auch für Privatpersonen/i.test(html)) add('Baustein doppelt: "Auch für Privatpersonen."', 'Block aus dem Artikeltext entfernen, er steht im Seitenlayout');
|
||||
if (/<h[23][^>]*>[^<]*(Muster anfordern|Handschriftmuster)/i.test(html)) add('Baustein doppelt: Muster-CTA', 'Abschnitt "Kostenloses Handschriftmuster anfordern." entfernen, er steht im Seitenlayout');
|
||||
// Querverweise auf andere Ratgeberartikel
|
||||
const quer = [...html.matchAll(/\[ratgeber_link\s+slug="([^"]+)"\]([\s\S]*?)\[\/ratgeber_link\]/g)];
|
||||
if (quer.length < 2) add('Querverweise: zu wenige', `${quer.length} [ratgeber_link], mindestens 2 auf verwandte Ratgeberartikel`);
|
||||
quer.filter(q => q[1] === artikel.slug).forEach(() => add('Querverweis auf sich selbst', artikel.slug));
|
||||
quer.filter(q => !q[2].trim()).forEach(q => add('Querverweis ohne Ankertext', q[1]));
|
||||
(html.match(/href="https?:\/\/(www\.)?skrift\.de\/(?!handgeschriebene-|unterschriftenservice|automatisierte-follow-ups|papier-schriftmuster|kontakt|handschrift-fuer)[a-z0-9-]+\/?"/g) || [])
|
||||
.forEach(h => add('Ratgeber-Link als harter Link', `${h} als [ratgeber_link slug="..."] setzen, sonst droht ein toter Link`));
|
||||
|
||||
const unbekannt = [...new Set(slugs.filter(s => s && !BUSINESS.includes(s) && !NEUTRAL.includes(s) && !PRIVAT.includes(s)))];
|
||||
unbekannt.forEach(s => add('Links: kein Linkziel fuer Ratgeber', `/${s}/ (z. B. Dankeseite, Rechtstexte)`));
|
||||
|
||||
|
||||
Reference in New Issue
Block a user