Loest die YouTube/Vimeo-API-Luecke: viele dienste laden ueber <script> (z. B. youtube.com/iframe_api, www-widgetapi.js, analytics) statt iframes - oft per JavaScript nachgeladen, daher fuer den scanner unsichtbar. - Pro dienst aktivierbar ueber das (umbenannte) feld "Zugehoerige Skripte blockieren (z. B. YouTube-/Vimeo-API)" = das vorhandene loads_script-flag. Presets (YouTube, Vimeo, Maps) haben es bereits an. - Server-seitig: passende <script src> werden zu type="text/plain" (src -> data-cb-src) neutralisiert, laden also nicht. - Client-seitig: winziger guard ganz frueh im <head> patcht appendChild/insertBefore/replaceChild und neutralisiert dynamisch injizierte scripts VOR dem einfuegen -> kein request. Faengt damit auch die per JS nachgeladene iframe_api ab. - Einwilligung (per-dienst-consent, z. B. ueber den video-platzhalter) schaltet die scripts via cbActivateScripts frei und laedt sie nach. - Neuer shortcode [content_blocker_consent id="…"] als einwilligungs-button fuer reine skript-dienste ohne sichtbaren platzhalter. - guard-logik mit DOM-mock getestet (block + reinject), server-regex isoliert geprueft. i18n DE/EN ergaenzt (127 strings). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
167 lines
5.3 KiB
PHP
167 lines
5.3 KiB
PHP
<?php
|
|
defined( 'ABSPATH' ) || exit;
|
|
|
|
/**
|
|
* Auto-detects iframes in the rendered HTML and replaces them with consent
|
|
* placeholders based on configured match patterns.
|
|
*
|
|
* Strategy (safe by design):
|
|
* 1. A regex LOCATES each <iframe>…</iframe> block (it does NOT parse attributes).
|
|
* 2. Each block is parsed individually with DOMDocument to read the src reliably.
|
|
* 3. Only the matched iframe block is replaced in the original HTML via callback.
|
|
*
|
|
* The rest of the page HTML is never re-serialized, so themes, page builders,
|
|
* inline JS and JSON-LD stay byte-for-byte intact.
|
|
*
|
|
* Known limitation: Only iframes present in the initial server-rendered HTML are
|
|
* covered here. For iframes injected later by JavaScript, use the manual shortcode
|
|
* [content_blocker id="…"] around the embed.
|
|
*/
|
|
class CB_Autodetect {
|
|
|
|
public static function init(): void {
|
|
add_action( 'template_redirect', [ __CLASS__, 'start_buffer' ] );
|
|
}
|
|
|
|
public static function start_buffer(): void {
|
|
if ( is_admin() || wp_doing_ajax() || wp_doing_cron() ) {
|
|
return;
|
|
}
|
|
if ( defined( 'REST_REQUEST' ) && REST_REQUEST ) {
|
|
return;
|
|
}
|
|
if ( defined( 'XMLRPC_REQUEST' ) && XMLRPC_REQUEST ) {
|
|
return;
|
|
}
|
|
|
|
$services = CB_Settings::get_services();
|
|
$active = array_values( array_filter(
|
|
$services,
|
|
fn( $s ) => ! empty( $s['match_pattern'] ) && ( $s['enabled'] ?? true )
|
|
) );
|
|
if ( empty( $active ) ) {
|
|
return;
|
|
}
|
|
|
|
ob_start( fn( string $html ) => self::process( $html, $active ) );
|
|
}
|
|
|
|
public static function process( string $html, array $services ): string {
|
|
if ( $html === '' ) {
|
|
return $html;
|
|
}
|
|
|
|
// Pass 1 — iframes: replace matched embeds with a consent placeholder.
|
|
if ( stripos( $html, '<iframe' ) !== false ) {
|
|
$out = preg_replace_callback(
|
|
'#<iframe\b[^>]*>.*?</iframe>#is',
|
|
function ( array $m ) use ( $services ): string {
|
|
return self::maybe_replace_iframe( $m[0], $services );
|
|
},
|
|
$html
|
|
);
|
|
// On a PCRE error (e.g. backtrack/recursion limit on a huge page),
|
|
// preg_replace_callback returns null. Never blank the page.
|
|
if ( $out !== null ) {
|
|
$html = $out;
|
|
}
|
|
}
|
|
|
|
// Pass 2 — scripts: neutralise <script src> of services that block scripts
|
|
// so they don't execute/fetch until consent (the JS guard re-injects them).
|
|
$script_services = array_values( array_filter(
|
|
$services,
|
|
static fn( $s ) => ! empty( $s['loads_script'] )
|
|
) );
|
|
if ( $script_services && stripos( $html, '<script' ) !== false ) {
|
|
$out = preg_replace_callback(
|
|
'#<script\b[^>]*\bsrc\s*=\s*["\'][^"\']+["\'][^>]*>#i',
|
|
function ( array $m ) use ( $script_services ): string {
|
|
return self::maybe_block_script( $m[0], $script_services );
|
|
},
|
|
$html
|
|
);
|
|
if ( $out !== null ) {
|
|
$html = $out;
|
|
}
|
|
}
|
|
|
|
return $html;
|
|
}
|
|
|
|
/**
|
|
* Neutralise a <script src="…"> opening tag if its src matches a script-
|
|
* blocking service: rename src→data-cb-src, force type="text/plain", and tag
|
|
* it with the service id. The early head guard re-injects it on consent.
|
|
*/
|
|
private static function maybe_block_script( string $tag, array $services ): string {
|
|
if ( ! preg_match( '/\bsrc\s*=\s*["\']([^"\']+)["\']/i', $tag, $m ) ) {
|
|
return $tag;
|
|
}
|
|
$src = str_replace( '&', '&', $m[1] );
|
|
|
|
foreach ( $services as $svc ) {
|
|
$pattern = $svc['match_pattern'] ?? '';
|
|
if ( $pattern === '' || ! str_contains( $src, $pattern ) ) {
|
|
continue;
|
|
}
|
|
$id = (string) ( $svc['id'] ?? '' );
|
|
// src → data-cb-src (first occurrence), drop any existing type, then
|
|
// force type="text/plain" + the service id on the opening tag.
|
|
$t = preg_replace( '#\bsrc(\s*=\s*)#i', 'data-cb-src$1', $tag, 1 );
|
|
$t = preg_replace( '#\btype\s*=\s*("[^"]*"|\'[^\']*\'|\S+)#i', '', $t );
|
|
$t = preg_replace(
|
|
'#^<script\b#i',
|
|
'<script type="text/plain" data-cb-id="' . esc_attr( $id ) . '"',
|
|
$t,
|
|
1
|
|
);
|
|
return $t ?? $tag;
|
|
}
|
|
return $tag;
|
|
}
|
|
|
|
/**
|
|
* Parse a single iframe block with DOMDocument, read its src, and return either
|
|
* the consent placeholder (on match) or the unchanged original block.
|
|
*/
|
|
private static function maybe_replace_iframe( string $iframe_html, array $services ): string {
|
|
$src = self::get_src( $iframe_html );
|
|
if ( $src === '' ) {
|
|
return $iframe_html;
|
|
}
|
|
|
|
foreach ( $services as $svc ) {
|
|
$pattern = $svc['match_pattern'] ?? '';
|
|
if ( $pattern !== '' && str_contains( $src, $pattern ) ) {
|
|
$attrs = CB_Renderer::extract_iframe_attrs( $iframe_html );
|
|
$dims = [ 'width' => $attrs['width'], 'height' => $attrs['height'] ];
|
|
return CB_Renderer::render_placeholder( $svc, $src, $dims );
|
|
}
|
|
}
|
|
|
|
return $iframe_html;
|
|
}
|
|
|
|
/** Reliably extract the src attribute of a single iframe via DOMDocument. */
|
|
private static function get_src( string $iframe_html ): string {
|
|
$dom = new DOMDocument();
|
|
libxml_use_internal_errors( true );
|
|
// Wrap so loadHTML has a clean context; force UTF-8 so umlauts survive.
|
|
$dom->loadHTML(
|
|
'<?xml encoding="utf-8"?><div>' . $iframe_html . '</div>',
|
|
LIBXML_HTML_NOIMPLIED | LIBXML_HTML_NODEFDTD
|
|
);
|
|
libxml_clear_errors();
|
|
|
|
$node = $dom->getElementsByTagName( 'iframe' )->item( 0 );
|
|
if ( ! $node instanceof DOMElement ) {
|
|
return '';
|
|
}
|
|
|
|
// getAttribute() already returns the entity-decoded value (e.g. & → &),
|
|
// so no further unescaping is needed here.
|
|
return trim( $node->getAttribute( 'src' ) );
|
|
}
|
|
}
|