Code Coverage
 
Lines
Functions and Methods
Classes and Traits
Total
76.13% covered (warning)
76.13%
169 / 222
60.00% covered (warning)
60.00%
15 / 25
CRAP
0.00% covered (danger)
0.00%
0 / 1
Render_Blocking_JS
76.13% covered (warning)
76.13%
169 / 222
60.00% covered (warning)
60.00%
15 / 25
200.22
0.00% covered (danger)
0.00%
0 / 1
 setup
100.00% covered (success)
100.00%
4 / 4
100.00% covered (success)
100.00%
1 / 1
1
 is_available
100.00% covered (success)
100.00%
1 / 1
100.00% covered (success)
100.00%
1 / 1
1
 register_data_sync
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
1
 get_change_output_action_names
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
1
 start_output_filtering
66.67% covered (warning)
66.67%
20 / 30
0.00% covered (danger)
0.00%
0 / 1
39.93
 handle_output_stream
100.00% covered (success)
100.00%
15 / 15
100.00% covered (success)
100.00%
1 / 1
6
 split_at_kept_scripts
68.18% covered (warning)
68.18%
15 / 22
0.00% covered (danger)
0.00%
0 / 1
5.81
 get_script_tags
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
2
 ignore_exclusion_scripts
100.00% covered (success)
100.00%
14 / 14
100.00% covered (success)
100.00%
1 / 1
2
 pin_position_dependent_scripts
100.00% covered (success)
100.00%
15 / 15
100.00% covered (success)
100.00%
1 / 1
5
 ignore_attribute_lookahead
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
1
 recalculate_buffer_split
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
1
 append_script_tags
87.50% covered (warning)
87.50%
7 / 8
0.00% covered (danger)
0.00%
0 / 1
3.02
 get_exclude_handles
100.00% covered (success)
100.00%
1 / 1
100.00% covered (success)
100.00%
1 / 1
1
 handle_exclusions
100.00% covered (success)
100.00%
5 / 5
100.00% covered (success)
100.00%
1 / 1
3
 has_dependents
75.00% covered (warning)
75.00%
3 / 4
0.00% covered (danger)
0.00%
0 / 1
3.14
 should_concatenate
100.00% covered (success)
100.00%
3 / 3
100.00% covered (success)
100.00%
1 / 1
3
 add_ignore_attribute
100.00% covered (success)
100.00%
1 / 1
100.00% covered (success)
100.00%
1 / 1
1
 is_opened_script
91.67% covered (success)
91.67%
11 / 12
0.00% covered (danger)
0.00%
0 / 1
2.00
 is_current_request_excluded
16.67% covered (danger)
16.67%
1 / 6
0.00% covered (danger)
0.00%
0 / 1
19.47
 is_url_excluded
33.33% covered (danger)
33.33%
3 / 9
0.00% covered (danger)
0.00%
0 / 1
12.41
 normalize_url_path
33.33% covered (danger)
33.33%
2 / 6
0.00% covered (danger)
0.00%
0 / 1
3.19
 strip_home_path
50.00% covered (danger)
50.00%
4 / 8
0.00% covered (danger)
0.00%
0 / 1
6.00
 get_exclusion_regex
56.25% covered (warning)
56.25%
18 / 32
0.00% covered (danger)
0.00%
0 / 1
15.78
 get_slug
100.00% covered (success)
100.00%
1 / 1
100.00% covered (success)
100.00%
1 / 1
1
1<?php
2/**
3 * Implements the system to avoid render blocking JS execution.
4 *
5 * @link       https://automattic.com
6 * @since      0.2
7 * @package    automattic/jetpack-boost
8 */
9
10namespace Automattic\Jetpack_Boost\Modules\Optimizations\Render_Blocking_JS;
11
12use Automattic\Jetpack\Schema\Schema;
13use Automattic\Jetpack\WP_JS_Data_Sync\Data_Sync;
14use Automattic\Jetpack_Boost\Contracts\Changes_Output_After_Activation;
15use Automattic\Jetpack_Boost\Contracts\Changes_Output_On_Activation;
16use Automattic\Jetpack_Boost\Contracts\Feature;
17use Automattic\Jetpack_Boost\Contracts\Has_Data_Sync;
18use Automattic\Jetpack_Boost\Contracts\Optimization;
19use Automattic\Jetpack_Boost\Data_Sync\Minify_Excludes_State_Entry;
20use Automattic\Jetpack_Boost\Lib\Body_Close_Locator;
21use Automattic\Jetpack_Boost\Lib\Output_Filter;
22
23/**
24 * Class Render_Blocking_JS
25 */
26class Render_Blocking_JS implements Feature, Changes_Output_On_Activation, Changes_Output_After_Activation, Optimization, Has_Data_Sync {
27    /**
28     * Substring that marks an inline script as producing position-dependent
29     * output (document.write()/document.writeln()). Such scripts must stay where
30     * they are rather than being moved to the end of the document. The fast-path
31     * guard and the per-script check must use the same needle to stay in lockstep.
32     *
33     * @var string
34     */
35    private const POSITION_DEPENDENT_OUTPUT_NEEDLE = 'document.write';
36
37    /**
38     * Holds the script tags removed from the output buffer.
39     *
40     * @var array
41     */
42    protected $buffered_script_tags = array();
43
44    /**
45     * `id` attributes of the tags printed for kept-in-place handles that other scripts depend on.
46     *
47     * @var string[]
48     */
49    private $kept_script_ids = array();
50
51    /**
52     * HTML attribute name to be added to <script> tag to make it
53     * ignored by this class.
54     *
55     * @var string|null
56     */
57    private $ignore_attribute;
58
59    /**
60     * HTML attribute value to be added to <script> tag to make it
61     * ignored by this class.
62     *
63     * @var string
64     */
65    private $ignore_value = 'ignore';
66
67    /**
68     * Utility class that supports output filtering.
69     *
70     * @var Output_Filter
71     */
72    private $output_filter = null;
73
74    /**
75     * Flag indicating an opened <script> tag in output.
76     *
77     * @var string
78     */
79    private $is_opened_script = false;
80
81    public function setup() {
82        $this->output_filter = new Output_Filter();
83
84        /**
85         * Filters the ignore attribute
86         *
87         * @param $string $ignore_attribute The string used to ignore elements of the page.
88         *
89         * @since   1.0.0
90         */
91        $this->ignore_attribute = apply_filters( 'jetpack_boost_render_blocking_js_ignore_attribute', 'data-jetpack-boost' );
92
93        add_action( 'template_redirect', array( $this, 'start_output_filtering' ), -999999 );
94
95        /**
96         * Shortcodes can sometimes output script to embed widget. It's safer to ignore them.
97         */
98        add_filter( 'do_shortcode_tag', array( $this, 'add_ignore_attribute' ) );
99    }
100
101    public static function is_available() {
102        return true;
103    }
104
105    /**
106     * Register the data sync entry holding the list of URL patterns
107     * excluded from JS deferring.
108     *
109     * @param Data_Sync $instance The data sync instance.
110     */
111    public function register_data_sync( Data_Sync $instance ) {
112        $instance->register(
113            'render_blocking_js_excludes',
114            Schema::as_array( Schema::as_string() )->fallback( array() ),
115            new Minify_Excludes_State_Entry( 'render_blocking_js_excludes' )
116        );
117    }
118
119    /**
120     * Cached pages need to be invalidated when the exclusion list changes.
121     *
122     * @return string[] Action names fired when the exclusion list is updated.
123     */
124    public static function get_change_output_action_names() {
125        $option = JETPACK_BOOST_DATASYNC_NAMESPACE . '_render_blocking_js_excludes';
126
127        // `add_option_*` covers the very first save, when the option is created
128        // rather than updated, so the cache is invalidated on that write too.
129        return array(
130            'add_option_' . $option,
131            'update_option_' . $option,
132        );
133    }
134
135    /**
136     * Set up an output filtering callback.
137     *
138     * @return void
139     */
140    public function start_output_filtering() {
141        /**
142         * We're doing heavy output filtering in this module
143         * by using output buffering.
144         *
145         * Here are a few scenarios when we shouldn't do it:
146         */
147
148        /**
149         * Filter to disable defer blocking JS
150         *
151         * @param bool $defer return false to disable defer blocking
152         *
153         * @since   1.0.0
154         */
155        if ( false === apply_filters( 'jetpack_boost_should_defer_js', '__return_true' ) ) {
156            return;
157        }
158
159        // Disable in robots.txt.
160        if ( isset( $_SERVER['REQUEST_URI'] ) && strpos( home_url( wp_unslash( $_SERVER['REQUEST_URI'] ) ), 'robots.txt' ) !== false ) { // phpcs:ignore WordPress.Security.ValidatedSanitizedInput.InputNotSanitized -- This is validating.
161            return;
162        }
163
164        // Disable in other possible AJAX requests setting cors related header.
165        if ( isset( $_SERVER['HTTP_SEC_FETCH_MODE'] ) && 'cors' === strtolower( $_SERVER['HTTP_SEC_FETCH_MODE'] ) ) { // phpcs:ignore WordPress.Security.ValidatedSanitizedInput -- This is validating.
166            return;
167        }
168
169        // Disable in other possible AJAX requests setting XHR related header.
170        if ( isset( $_SERVER['HTTP_X_REQUESTED_WITH'] ) && 'xmlhttprequest' === strtolower( $_SERVER['HTTP_X_REQUESTED_WITH'] ) ) { // phpcs:ignore WordPress.Security.ValidatedSanitizedInput -- This is validating.
171            return;
172        }
173
174        // Disable in all XLS (see the WP_Sitemaps_Renderer class which is responsible for rendering Sitemaps data to XML
175        // in accordance with sitemap protocol).
176        if ( isset( $_SERVER['REQUEST_URI'] ) &&
177            (
178                // phpcs:disable WordPress.Security.ValidatedSanitizedInput -- This is validating.
179                str_contains( $_SERVER['REQUEST_URI'], '.xsl' ) ||
180                str_contains( $_SERVER['REQUEST_URI'], 'sitemap-stylesheet=index' ) ||
181                str_contains( $_SERVER['REQUEST_URI'], 'sitemap-stylesheet=sitemap' )
182                // phpcs:enable WordPress.Security.ValidatedSanitizedInput
183            ) ) {
184            return;
185        }
186
187        // Disable in all POST Requests.
188        // phpcs:disable WordPress.Security.NonceVerification.Missing
189        if ( ! empty( $_POST ) ) {
190            return;
191        }
192
193        // Disable in customizer previews
194        if ( is_customize_preview() ) {
195            return;
196        }
197
198        // Disable in feeds, AJAX, Cron, XML.
199        if ( is_feed() || wp_doing_ajax() || wp_doing_cron() || wp_is_xml_request() ) {
200            return;
201        }
202
203        // Disable in sitemaps.
204        if ( ! empty( get_query_var( 'sitemap' ) ) ) {
205            return;
206        }
207
208        // Disable in AMP pages.
209        if ( function_exists( 'amp_is_request' ) && amp_is_request() ) {
210            return;
211        }
212
213        // Disable on URLs excluded by the user.
214        if ( $this->is_current_request_excluded() ) {
215            // Leave the page output completely untouched, as if the module was off.
216            remove_filter( 'do_shortcode_tag', array( $this, 'add_ignore_attribute' ) );
217            return;
218        }
219
220        // Print the filtered script tags to the very end of the page.
221        add_filter( 'jetpack_boost_output_filtering_last_buffer', array( $this, 'append_script_tags' ), 10, 1 );
222
223        // Handle exclusions.
224        add_filter( 'script_loader_tag', array( $this, 'handle_exclusions' ), 10, 2 );
225        add_filter( 'js_do_concat', array( $this, 'should_concatenate' ), 10, 2 );
226
227        $this->output_filter->add_callback( array( $this, 'handle_output_stream' ) );
228    }
229
230    /**
231     * Remove all inline and external <script> tags from the default output.
232     *
233     * @param string $buffer_start First part of the buffer.
234     * @param string $buffer_end   Second part of the buffer.
235     *
236     * For explanation on why there are two parts of a buffer here, see
237     * the comments and examples in the Output_Filter class.
238     *
239     * @return array Parts of the buffer.
240     */
241    public function handle_output_stream( $buffer_start, $buffer_end ) {
242        $joint_buffer = $this->ignore_exclusion_scripts( $buffer_start . $buffer_end );
243
244        list( $kept_in_place, $joint_buffer ) = $this->split_at_kept_scripts( $joint_buffer );
245
246        $script_tags = $this->get_script_tags( $joint_buffer );
247
248        if ( ! $script_tags ) {
249            if ( '' !== $kept_in_place ) {
250                // A script left open by an earlier chunk may have closed in the part printed in place.
251                $this->is_opened_script = $this->is_opened_script( $joint_buffer );
252            }
253
254            // We have an opened script tag, move everything to the second buffer to avoid printing it to the page.
255            // We will do this until the </script> closing tag is encountered.
256            if ( $this->is_opened_script || '' !== $kept_in_place ) {
257                return array( $kept_in_place, $joint_buffer );
258            }
259
260            // No script tags detected, return both chunks unaltered.
261            return array( $buffer_start, $buffer_end );
262        }
263
264        // Makes sure all whole <script>...</script> tags are in $buffer_start.
265        list( $buffer_start, $buffer_end ) = $this->recalculate_buffer_split( $joint_buffer, $script_tags );
266
267        foreach ( $script_tags as $script_tag ) {
268            $this->buffered_script_tags[] = $script_tag[0];
269            $buffer_start                 = str_replace( $script_tag[0], '', $buffer_start );
270        }
271
272        // Detect a lingering opened script.
273        $this->is_opened_script = $this->is_opened_script( $buffer_start . $buffer_end );
274
275        return array( $kept_in_place . $buffer_start, $buffer_end );
276    }
277
278    /**
279     * Keep every script up to the last kept-in-place library in document order.
280     *
281     * Scripts printed before a library may set up state around it, like Divi's inline jQuery
282     * stand-in before `jquery-core` (BOOST-763), so moving them after it breaks the page.
283     *
284     * @param string $buffer Captured piece of output buffer.
285     *
286     * @return string[] The part to print unchanged, and the rest of the buffer.
287     */
288    private function split_at_kept_scripts( $buffer ) {
289        if ( ! $this->kept_script_ids ) {
290            return array( '', $buffer );
291        }
292
293        $ids   = array_map(
294            function ( $id ) {
295                return preg_quote( $id, '~' );
296            },
297            array_unique( $this->kept_script_ids )
298        );
299        $regex = '~<script\b[^>]*\sid=(["\'])(?:' . implode( '|', $ids ) . ')\1[^>]*>[\s\S]*?</script>~i';
300
301        if ( ! preg_match_all( $regex, $buffer, $kept_tags, PREG_OFFSET_CAPTURE ) ) {
302            return array( '', $buffer );
303        }
304
305        $last_kept = end( $kept_tags[0] );
306        $split_at  = $last_kept[1] + strlen( $last_kept[0] );
307        $kept      = substr( $buffer, 0, $split_at );
308
309        // Scripts taken from earlier chunks go back in front of the first script here, where they were printed.
310        if ( $this->buffered_script_tags ) {
311            $first_script = $kept_tags[0][0][1];
312            $moved_here   = $this->get_script_tags( $kept );
313            if ( $moved_here ) {
314                $first_script = min( $first_script, $moved_here[0][1] );
315            }
316
317            $kept                       = substr_replace( $kept, implode( '', $this->buffered_script_tags ), $first_script, 0 );
318            $this->buffered_script_tags = array();
319        }
320
321        return array( $kept, substr( $buffer, $split_at ) );
322    }
323
324    /**
325     * Matches <script> tags with their content in a string buffer.
326     *
327     * @param string $buffer Captured piece of output buffer.
328     *
329     * @return array
330     */
331    protected function get_script_tags( $buffer ) {
332        $regex = '~<script' . $this->ignore_attribute_lookahead() . '([^>]*)>[\s\S]*?<\/script>~si';
333        preg_match_all( $regex, $buffer, $script_tags, PREG_OFFSET_CAPTURE );
334
335        // No script_tags in the joint buffer.
336        if ( empty( $script_tags[0] ) ) {
337            return array();
338        }
339
340        /**
341         * Filter to remove any scripts that should not be moved to the end of the document.
342         *
343         * @param array $script_tags array of script tags. Remove any scripts that should not be moved to the end of the documents.
344         *
345         * @since   1.0.0
346         */
347        return apply_filters( 'jetpack_boost_render_blocking_js_exclude_scripts', $script_tags[0] );
348    }
349
350    /**
351     * Adds the ignore attribute to scripts in the exclusion list.
352     *
353     * @param string $buffer Captured piece of output buffer.
354     *
355     * @return string
356     */
357    protected function ignore_exclusion_scripts( $buffer ) {
358        $exclusions = array(
359            // Scripts inside HTML comments.
360            '~<!--.*?-->~si',
361
362            // Scripts with types that do not execute complex code. Moving them down can be dangerous
363            // and does not benefit performance. Includes types: application/json, application/ld+json and importmap.
364            '~<script\s+[^\>]*type=(?<q>["\']*)(application\/(ld\+)?json|importmap)\k<q>.*?>.*?<\/script>~si',
365        );
366
367        $excluded = preg_replace_callback(
368            $exclusions,
369            function ( $script_match ) {
370                return $this->add_ignore_attribute( $script_match[0] );
371            },
372            $buffer
373        );
374        // preg_replace_callback() returns null on PCRE failure; keep the original
375        // buffer in that case rather than propagating null downstream.
376        if ( null !== $excluded ) {
377            $buffer = $excluded;
378        }
379
380        return $this->pin_position_dependent_scripts( $buffer );
381    }
382
383    /**
384     * Keep inline scripts whose output is position-dependent in their original place.
385     *
386     * Scripts using document.write()/document.writeln() insert markup at the script's
387     * location, so moving such a script to the end of the document renders its output
388     * after the footer instead of inside the content (e.g. a Custom HTML block).
389     * Marking the script with the ignore attribute keeps the rest of the pipeline
390     * from moving it. Scripts that already carry the ignore attribute are skipped so
391     * their behavior and markup are unchanged.
392     *
393     * Best-effort and deliberately conservative: it pins the common case (an inline
394     * script that calls document.write) and otherwise leaves the script to the
395     * default move behavior. It does not pin scripts that write their own
396     * '<script ...>' markup (no safe in-place edit exists), nor exotic call forms a
397     * substring check cannot see. A miss never corrupts the page â€” worst case is a
398     * script that still moves, exactly as it does without this method.
399     *
400     * @param string $buffer Captured piece of output buffer.
401     *
402     * @return string
403     */
404    private function pin_position_dependent_scripts( $buffer ) {
405        // Fast path: skip the inline-script scan entirely when the buffer cannot
406        // contain a position-dependent script.
407        if ( false === stripos( $buffer, self::POSITION_DEPENDENT_OUTPUT_NEEDLE ) ) {
408            return $buffer;
409        }
410
411        // Match inline scripts only (no src attribute) that do not already carry
412        // the ignore attribute. This runs on the buffer Output_Filter hands us
413        // (a bounded window), not a whole page; the lazy [\s\S]*? would be
414        // superlinear on a multi-megabyte single buffer, so keep it window-scoped.
415        $inline_script_regex = '~<script\b(?![^>]*\ssrc\s*=)' . $this->ignore_attribute_lookahead() . '[^>]*>[\s\S]*?</script>~i';
416
417        $result = preg_replace_callback(
418            $inline_script_regex,
419            function ( $script_match ) {
420                // Intentionally conservative: a simple case-insensitive substring check
421                // for "document.write" (which also covers "document.writeln"). It does
422                // not parse JS, so exotic call forms it cannot see â€” document['write'](),
423                // "document . write()", or an uppercase <SCRIPT> tag that
424                // add_ignore_attribute()'s lowercase replace won't touch â€” simply fall
425                // back to the default behavior (the script is moved, as it is today).
426                // That is the safe direction: a miss never corrupts the page.
427                if ( false === stripos( $script_match[0], self::POSITION_DEPENDENT_OUTPUT_NEEDLE ) ) {
428                    return $script_match[0];
429                }
430
431                // Do not touch a script that writes its own '<script ...>' markup. There
432                // is no safe in-place edit for it: add_ignore_attribute() does a global
433                // str_replace() on '<script', which rewrites the inner literal and can
434                // break the quoting of the string the script writes; tagging only the
435                // outer tag would instead let get_script_tags() match and move that inner
436                // literal. Such scripts keep the default behavior rather than risk
437                // corrupting the page.
438                if ( substr_count( strtolower( $script_match[0] ), '<script' ) > 1 ) {
439                    return $script_match[0];
440                }
441
442                return $this->add_ignore_attribute( $script_match[0] );
443            },
444            $buffer
445        );
446
447        // preg_replace_callback() returns null on PCRE failure (e.g. backtrack limit
448        // on a pathological buffer); fall back to the unmodified buffer so the page is
449        // never blanked. Mirrors the guard in is_opened_script().
450        return null === $result ? $buffer : $result;
451    }
452
453    /**
454     * Negative lookahead asserting a <script> tag does not already carry the
455     * ignore attribute. Shared by the regexes that select movable scripts so the
456     * attribute-matching rule lives in one place.
457     *
458     * @return string Regex fragment (uses named group "q"; safe to use once per pattern).
459     */
460    private function ignore_attribute_lookahead() {
461        return sprintf(
462            '(?![^>]*%s=(?<q>["\']*)%s\k<q>)',
463            preg_quote( $this->ignore_attribute, '~' ),
464            preg_quote( $this->ignore_value, '~' )
465        );
466    }
467
468    /**
469     * Splits the buffer into two parts.
470     *
471     * First part contains all whole <script> tags, the second part
472     * contains the rest of the buffer.
473     *
474     * @param string $buffer      Captured piece of output buffer.
475     * @param array  $script_tags Matched <script> tags.
476     *
477     * @return array
478     */
479    protected function recalculate_buffer_split( $buffer, $script_tags ) {
480        $last_script_tag_index        = count( $script_tags ) - 1;
481        $last_script_tag_end_position = strrpos( $buffer, $script_tags[ $last_script_tag_index ][0] ) + strlen( $script_tags[ $last_script_tag_index ][0] );
482
483        // Bundle all script tags into the first buffer.
484        $buffer_start = substr( $buffer, 0, $last_script_tag_end_position );
485
486        // Leave the rest of the data in the second buffer.
487        $buffer_end = substr( $buffer, $last_script_tag_end_position );
488
489        return array( $buffer_start, $buffer_end );
490    }
491
492    /**
493     * Insert the buffered script tags just before the document's closing body
494     * tag if the last buffer holds one, otherwise append them at the end.
495     *
496     * The closing tag is located with an HTML tokenizer rather than string
497     * search: a literal '</body>' inside a script's source (document.write),
498     * a textarea, a comment or an attribute value is content, not markup, and
499     * inserting there corrupts the page (BOOST-585).
500     *
501     * @param string $buffer String buffer.
502     *
503     * @return string
504     */
505    public function append_script_tags( $buffer ) {
506        $script_tags = implode( '', $this->buffered_script_tags );
507        // Reset tags in case there's another buffer after this one.
508        $this->buffered_script_tags = array();
509
510        // Nothing to insert: both branches below are identity operations, so
511        // skip the buffer scan entirely. Any other feature registering an
512        // Output_Filter on the same global hook â€” Lcp does â€” calls this a
513        // second time per request.
514        if ( '' === $script_tags ) {
515            return $buffer;
516        }
517
518        $position = Body_Close_Locator::find( $buffer );
519        if ( null === $position ) {
520            return $buffer . $script_tags;
521        }
522
523        return substr_replace( $buffer, $script_tags, $position, 0 );
524    }
525
526    /**
527     * Handles that must keep their place in the document, as provided by
528     * `jetpack_boost_render_blocking_js_exclude_handles`.
529     *
530     * @return array
531     */
532    private function get_exclude_handles() {
533        /**
534         * Filter to provide an array of registered script handles that should not be moved to the end of the document.
535         *
536         * @param array $script_handles array of script handles. Remove any scripts that should not be moved to the end of the documents.
537         *
538         * @since   1.0.0
539         */
540        return (array) apply_filters( 'jetpack_boost_render_blocking_js_exclude_handles', array() );
541    }
542
543    /**
544     * Exclude certain scripts from being processed by this class.
545     *
546     * @param string $tag    <script> opening tag.
547     * @param string $handle Script handle from register_ or enqueue_ methods.
548     *
549     * @return string
550     */
551    public function handle_exclusions( $tag, $handle ) {
552        if ( ! in_array( $handle, $this->get_exclude_handles(), true ) ) {
553            return $tag;
554        }
555
556        // A kept script nothing depends on, like the Likes queue handler in the footer, needs no
557        // scripts before it kept in place, and making it a barrier would stop deferral on the whole page.
558        if ( $this->has_dependents( $handle ) ) {
559            $this->kept_script_ids[] = $handle . '-js';
560        }
561
562        return $this->add_ignore_attribute( $tag );
563    }
564
565    /**
566     * Whether any registered script depends on the handle.
567     *
568     * @param string $handle Script handle.
569     *
570     * @return bool
571     */
572    private function has_dependents( $handle ) {
573        foreach ( wp_scripts()->registered as $script ) {
574            if ( in_array( $handle, (array) $script->deps, true ) ) {
575                return true;
576            }
577        }
578
579        return false;
580    }
581
582    /**
583     * Whether Minify JS may concatenate a script, given the handles excluded from deferral.
584     *
585     * Concatenated scripts share one <script> tag, but handle_exclusions() marks a script's own
586     * tag - a concatenated script has none to mark, so this module would move it.
587     *
588     * @param mixed  $do_concat Whether the script may be concatenated, as left by earlier filters.
589     * @param string $handle    Script handle from register_ or enqueue_ methods.
590     *
591     * @return mixed False when this module vetoes, otherwise $do_concat unchanged.
592     */
593    public function should_concatenate( $do_concat, $handle ) {
594        if ( $do_concat && in_array( $handle, $this->get_exclude_handles(), true ) ) {
595            return false;
596        }
597
598        // Not a fresh boolean: Concatenate_JS concatenates only on `true === $do_concat`.
599        return $do_concat;
600    }
601
602    /**
603     * Add the ignore attribute to the script tags.
604     *
605     * Case-insensitive so uppercase/mixed-case tags (`<SCRIPT>`, valid HTML and
606     * common in hand-written Custom HTML / legacy embeds) are handled too; a
607     * case-sensitive match would silently no-op on them and leave them movable.
608     *
609     * @param string $html HTML code possibly containing a <script> opening tag.
610     *
611     * @return string
612     */
613    public function add_ignore_attribute( $html ) {
614        return str_ireplace( '<script', sprintf( '<script %s="%s"', esc_html( $this->ignore_attribute ), esc_attr( $this->ignore_value ) ), $html );
615    }
616
617    /**
618     * Detects an unclosed script tag in a buffer.
619     *
620     * @param string $buffer Joint buffer.
621     *
622     * @return bool
623     */
624    public function is_opened_script( $buffer ) {
625        // Strip fully-paired ignored <script>...</script> blocks so the counts below are symmetric.
626        $ignored_pair_regex = sprintf(
627            '~<script[^>]*%s=(?<q>["\']*)%s\k<q>[^>]*>[\s\S]*?</script>~si',
628            preg_quote( $this->ignore_attribute, '~' ),
629            preg_quote( $this->ignore_value, '~' )
630        );
631        $stripped           = preg_replace( $ignored_pair_regex, '', $buffer );
632        if ( null === $stripped ) {
633            $stripped = $buffer;
634        }
635
636        // Strip HTML comments so a commented-out </script> doesn't skew the count.
637        $stripped = preg_replace( '~<!--[\s\S]*?-->~', '', $stripped ) ?? $stripped;
638
639        $opening_tags_count = preg_match_all( '~<\s*script(\s[^>]*)?>~i', $stripped );
640        $closing_tags_count = preg_match_all( '~<\s*/\s*script\s*>~i', $stripped );
641
642        return $opening_tags_count > $closing_tags_count;
643    }
644
645    /**
646     * Checks if the current request URL matches one of the exclusion patterns
647     * configured by the user.
648     *
649     * Runs at template_redirect time, when REQUEST_URI is available.
650     *
651     * @return bool
652     */
653    private function is_current_request_excluded() {
654        if ( ! isset( $_SERVER['REQUEST_URI'] ) ) {
655            return false;
656        }
657
658        $patterns = function_exists( 'jetpack_boost_ds_get' ) ? jetpack_boost_ds_get( 'render_blocking_js_excludes' ) : array();
659        if ( empty( $patterns ) || ! is_array( $patterns ) ) {
660            return false;
661        }
662
663        // phpcs:ignore WordPress.Security.ValidatedSanitizedInput.InputNotSanitized -- Only used for comparison.
664        return self::is_url_excluded( wp_unslash( $_SERVER['REQUEST_URI'] ), $patterns );
665    }
666
667    /**
668     * Checks whether a request URI matches any of the given exclusion patterns.
669     *
670     * Patterns follow the semantics documented for Page Cache bypass patterns:
671     * they are compared against the URL path (query strings are ignored),
672     * a `(.*)` or `*` wildcard matches any part of the path, trailing slashes
673     * are optional and the comparison is case-insensitive.
674     *
675     * Two things differ from Page Cache, so keep them in mind before unifying the
676     * two implementations: every character outside the wildcard tokens is escaped
677     * via preg_quote() and matched literally (a pattern like `page.html` never
678     * acts as a regular expression), and the path is percent-decoded so a pattern
679     * typed as it appears in the address bar matches an encoded request path.
680     *
681     * @param string $request_uri The request URI to check.
682     * @param array  $patterns    List of URL patterns.
683     *
684     * @return bool
685     */
686    public static function is_url_excluded( $request_uri, $patterns ) {
687        $path = self::normalize_url_path( $request_uri );
688
689        foreach ( $patterns as $pattern ) {
690            $regex = self::get_exclusion_regex( $pattern );
691            if ( null === $regex ) {
692                continue;
693            }
694
695            $matched = preg_match( $regex, $path );
696
697            /*
698             * preg_match() returns false when PCRE cannot evaluate the pattern â€”
699             * e.g. a pathological pattern with several literal-separated wildcards
700             * hits the backtrack limit on a long URL. Treat that as a match so a
701             * deliberate exclusion is honoured (defer stays off on the page)
702             * rather than silently ignored.
703             */
704            if ( 1 === $matched || false === $matched ) {
705                return true;
706            }
707        }
708
709        return false;
710    }
711
712    /**
713     * Extracts a normalized path from a URL or request URI.
714     *
715     * Drops the query string, ensures a leading slash and removes trailing
716     * slashes (except for the root path).
717     *
718     * @param string $url URL or request URI.
719     *
720     * @return string
721     */
722    private static function normalize_url_path( $url ) {
723        $path = (string) wp_parse_url( $url, PHP_URL_PATH );
724
725        // Decode percent-encoding so a pattern typed as it appears in the address
726        // bar (e.g. `foo bar`, or a non-ASCII slug) matches the encoded request
727        // path (`/foo%20bar`). Both the pattern and the request pass through here,
728        // so the two sides stay symmetric.
729        $path = rawurldecode( $path );
730
731        $path = '/' . ltrim( $path, '/' );
732
733        if ( '/' !== $path ) {
734            $path = rtrim( $path, '/' );
735        }
736
737        return self::strip_home_path( $path );
738    }
739
740    /**
741     * Removes the site's home directory prefix from a path.
742     *
743     * On a subdirectory install (e.g. a site at `/blog/`) the request URI
744     * includes the subdirectory but user-entered patterns generally do not.
745     * Stripping the home directory from both sides makes the comparison relative
746     * to the home root, so a `checkout` pattern matches `/blog/checkout`.
747     *
748     * @param string $path A normalized URL path (leading slash, no query/trailing slash).
749     *
750     * @return string
751     */
752    private static function strip_home_path( $path ) {
753        $home_path = rtrim( (string) wp_parse_url( home_url( '/' ), PHP_URL_PATH ), '/' );
754
755        if ( '' === $home_path ) {
756            return $path;
757        }
758
759        if ( 0 === strcasecmp( $path, $home_path ) ) {
760            return '/';
761        }
762
763        if ( 0 === strncasecmp( $path, $home_path . '/', strlen( $home_path ) + 1 ) ) {
764            return substr( $path, strlen( $home_path ) );
765        }
766
767        return $path;
768    }
769
770    /**
771     * Turns a single exclusion pattern into an anchored regular expression.
772     *
773     * @param mixed $pattern A user-provided URL pattern.
774     *
775     * @return string|null The regular expression, or null if the pattern is empty.
776     */
777    private static function get_exclusion_regex( $pattern ) {
778        if ( ! is_string( $pattern ) ) {
779            return null;
780        }
781
782        $pattern = trim( $pattern );
783        if ( '' === $pattern ) {
784            return null;
785        }
786
787        /*
788         * Reject malformed URL patterns. A full URL with a scheme but no path
789         * (e.g. a typo'd `http://[::1`) would otherwise collapse to `/` and
790         * silently exclude only the homepage. A pathless URL that points at this
791         * site (e.g. the home URL pasted as `https://example.com`) is allowed
792         * through, since it legitimately means the homepage.
793         */
794        $parsed = wp_parse_url( $pattern );
795        if ( false === $parsed ) {
796            return null;
797        }
798        if ( isset( $parsed['scheme'] ) && empty( $parsed['path'] ) ) {
799            $home_host = wp_parse_url( home_url( '/' ), PHP_URL_HOST );
800            if ( empty( $parsed['host'] ) || 0 !== strcasecmp( $parsed['host'], (string) $home_host ) ) {
801                return null;
802            }
803        }
804
805        // Allow full URLs by stripping the home URL prefix (both secure and non-secure).
806        $home_url = home_url( '/' );
807        $pattern  = str_ireplace(
808            array(
809                $home_url,
810                str_replace( 'http:', 'https:', $home_url ),
811            ),
812            '/',
813            $pattern
814        );
815
816        $pattern = self::normalize_url_path( $pattern );
817
818        /*
819         * Split on wildcard tokens, treating any run of adjacent wildcards as a
820         * single split point. The possessive `++` is important: without coalescing,
821         * a pattern such as `****` would expand to one `.*` group per character, and
822         * thousands of wildcards would compile to thousands of groups and exhaust
823         * memory when matched against every front-end request. Possessive (rather
824         * than greedy `+`) keeps the split itself linear, so a pathological run of
825         * thousands of adjacent wildcards cannot exhaust the PCRE backtrack/JIT
826         * stack and make preg_split() return false.
827         */
828        $tokens = preg_split( '/(?:\(\.\*\)|\(\*\)|\.\*|\*)++/', $pattern );
829        if ( false === $tokens ) {
830            return null;
831        }
832
833        // Everything between wildcards is matched literally; only the wildcards
834        // become a (non-capturing) `.*` group.
835        $quoted = array_map(
836            function ( $token ) {
837                return preg_quote( $token, '~' );
838            },
839            $tokens
840        );
841
842        return '~^' . implode( '(?:.*)', $quoted ) . '/?$~i';
843    }
844
845    public static function get_slug() {
846        return 'render_blocking_js';
847    }
848}