#!/usr/bin/env python3 # -*- coding: utf-8 -*- """Build gdpr-content-blocker-en_US.po + .mo. Robust and order-independent: every German source string (extracted into hilfsdaten/strings.txt) is looked up in hilfsdaten/translations-en.json. Adding a new __() string to the plugin only requires adding one key/value to that JSON file — order does not matter. The build FAILS (and lists them) if any extracted string has no translation. Usage: python hilfsdaten/extract-strings.py gdpr-content-blocker hilfsdaten/strings.txt python hilfsdaten/build-en-mo.py . """ import json import os import struct import sys base = sys.argv[1] if len(sys.argv) > 1 else '.' here = os.path.dirname(os.path.abspath(__file__)) with open(os.path.join(here, 'translations-en.json'), encoding='utf-8') as f: TRANSLATIONS = json.load(f) strings_file = os.path.join(base, 'hilfsdaten', 'strings.txt') de = [] with open(strings_file, encoding='utf-8') as f: for line in f: if line.startswith('MSGID\t'): de.append(line[len('MSGID\t'):].rstrip('\n')) missing = [s for s in de if s not in TRANSLATIONS] if missing: print('MISSING %d translation(s) — add them to hilfsdaten/translations-en.json:' % len(missing)) for m in missing: print(' ' + json.dumps(m, ensure_ascii=False)) sys.exit(1) header = ( 'Project-Id-Version: GDPR Content Blocker\n' 'MIME-Version: 1.0\n' 'Content-Type: text/plain; charset=UTF-8\n' 'Content-Transfer-Encoding: 8bit\n' 'Language: en_US\n' ) entries = {'': header} for s in de: entries[s] = TRANSLATIONS[s] def po_escape(s): return s.replace('\\', '\\\\').replace('"', '\\"').replace('\n', '\\n') po_path = os.path.join(base, 'gdpr-content-blocker', 'languages', 'gdpr-content-blocker-en_US.po') with open(po_path, 'w', encoding='utf-8') as f: f.write('msgid ""\nmsgstr ""\n') for line in header.rstrip('\n').split('\n'): f.write('"%s\\n"\n' % po_escape(line)) f.write('\n') for s in de: f.write('msgid "%s"\n' % po_escape(s)) f.write('msgstr "%s"\n\n' % po_escape(TRANSLATIONS[s])) def write_mo(entries, out): keys = sorted(entries.keys()) ids = b'' strs = b'' offsets = [] for k in keys: kb = k.encode('utf-8') vb = entries[k].encode('utf-8') offsets.append((len(ids), len(kb), len(strs), len(vb))) ids += kb + b'\x00' strs += vb + b'\x00' keystart = 7 * 4 + 16 * len(keys) valuestart = keystart + len(ids) ko = [] vo = [] for o1, l1, o2, l2 in offsets: ko += [l1, o1 + keystart] vo += [l2, o2 + valuestart] out_b = struct.pack('Iiiiiii', 0x950412de, 0, len(keys), 7 * 4, 7 * 4 + len(keys) * 8, 0, 0) out_b += struct.pack('i' * len(ko), *ko) + struct.pack('i' * len(vo), *vo) + ids + strs with open(out, 'wb') as f: f.write(out_b) mo_path = os.path.join(base, 'gdpr-content-blocker', 'languages', 'gdpr-content-blocker-en_US.mo') write_mo(entries, mo_path) print('OK: %d strings -> %s + .mo' % (len(de), os.path.basename(po_path)))