• Home
  • Features
  • Pricing
  • Docs
  • Announcements
  • Sign In

MyIntervals / emogrifier / 22869219019

09 Mar 2026 06:44PM UTC coverage: 96.203% (-0.1%) from 96.305%
22869219019

Pull #1588

github

web-flow
Merge c9816373d into 252b16b8e
Pull Request #1588: [BUGFIX] Throw exception if `DOMDocument::saveHTML` fails

6 of 7 new or added lines in 1 file covered. (85.71%)

836 of 869 relevant lines covered (96.2%)

259.7 hits per line

Source File
Press 'n' to go to next uncovered line, 'b' for previous

90.23
/src/HtmlProcessor/AbstractHtmlProcessor.php
1
<?php
2

3
declare(strict_types=1);
4

5
namespace Pelago\Emogrifier\HtmlProcessor;
6

7
use function Safe\preg_match;
8
use function Safe\preg_replace;
9

10
/**
11
 * Base class for HTML processor that e.g., can remove, add or modify nodes or attributes.
12
 *
13
 * The "vanilla" subclass is the HtmlNormalizer.
14
 */
15
abstract class AbstractHtmlProcessor
16
{
17
    protected const DEFAULT_DOCUMENT_TYPE = '<!DOCTYPE html>';
18
    protected const CONTENT_TYPE_META_TAG = '<meta http-equiv="Content-Type" content="text/html; charset=utf-8">';
19

20
    /**
21
     * Regular expression part to match tag names that PHP's DOMDocument implementation is not
22
     * aware are self-closing. These are mostly HTML5 elements, but for completeness `<command>` (obsolete) and
23
     * `<keygen>` (deprecated) are also included.
24
     *
25
     * @see https://bugs.php.net/bug.php?id=73175
26
     */
27
    protected const PHP_UNRECOGNIZED_VOID_TAGNAME_MATCHER = '(?:command|embed|keygen|source|track|wbr)';
28

29
    /**
30
     * Regular expression part to match tag names that may appear before the start of the `<body>` element.  A start tag
31
     * for any other element would implicitly start the `<body>` element due to tag omission rules.
32
     */
33
    protected const TAGNAME_ALLOWED_BEFORE_BODY_MATCHER
34
        = '(?:html|head|base|command|link|meta|noscript|script|style|template|title)';
35

36
    /**
37
     * regular expression pattern to match an HTML comment, including delimiters and modifiers
38
     */
39
    protected const HTML_COMMENT_PATTERN = '/<!--[^-]*+(?:-(?!->)[^-]*+)*+(?:-->|$)/';
40

41
    /**
42
     * regular expression pattern to match an HTML `<template>` element, including delimiters and modifiers
43
     */
44
    protected const HTML_TEMPLATE_ELEMENT_PATTERN
45
        = '%<template[\\s>][^<]*+(?:<(?!/template>)[^<]*+)*+(?:</template>|$)%i';
46

47
    /**
48
     * @var \DOMDocument|null
49
     */
50
    protected $domDocument = null;
51

52
    /**
53
     * @var \DOMXPath|null
54
     */
55
    private $xPath = null;
56

57
    /**
58
     * The constructor.
59
     *
60
     * Please use `::fromHtml` or `::fromDomDocument` instead.
61
     */
62
    final private function __construct() {}
63

64
    /**
65
     * Builds a new instance from the given HTML.
66
     *
67
     * @param non-empty-string $unprocessedHtml raw HTML, must be UTF-encoded
68
     *
69
     * @return static
70
     *
71
     * @throws \InvalidArgumentException if $unprocessedHtml is anything other than a non-empty string
72
     */
73
    public static function fromHtml(string $unprocessedHtml): self
612 ✔
74
    {
75
        // @phpstan-ignore-next-line argument.type We're checking for a contract violation here.
76
        if ($unprocessedHtml === '') {
612 ✔
77
            throw new \InvalidArgumentException('The provided HTML must not be empty.', 1515763647);
1 ✔
78
        }
79

80
        $instance = new static();
611 ✔
81
        $instance->setHtml($unprocessedHtml);
611 ✔
82

83
        return $instance;
611 ✔
84
    }
85

86
    /**
87
     * Builds a new instance from the given DOM document.
88
     *
89
     * @param \DOMDocument $document a DOM document returned by getDomDocument() of another instance
90
     *
91
     * @return static
92
     */
93
    public static function fromDomDocument(\DOMDocument $document): self
4 ✔
94
    {
95
        $instance = new static();
4 ✔
96
        $instance->setDomDocument($document);
4 ✔
97

98
        return $instance;
4 ✔
99
    }
100

101
    /**
102
     * Sets the HTML to process.
103
     *
104
     * @param string $html the HTML to process, must be UTF-8-encoded
105
     */
106
    private function setHtml(string $html): void
611 ✔
107
    {
108
        $this->createUnifiedDomDocument($html);
611 ✔
109
    }
110

111
    /**
112
     * Provides access to the internal DOMDocument representation of the HTML in its current state.
113
     *
114
     * @throws \UnexpectedValueException
115
     */
116
    public function getDomDocument(): \DOMDocument
613 ✔
117
    {
118
        if (!$this->domDocument instanceof \DOMDocument) {
613 ✔
119
            $message = self::class . '::setDomDocument() has not yet been called on ' . static::class;
×
120
            throw new \UnexpectedValueException($message, 1570472239);
×
121
        }
122

123
        return $this->domDocument;
613 ✔
124
    }
125

126
    private function setDomDocument(\DOMDocument $domDocument): void
615 ✔
127
    {
128
        $this->domDocument = $domDocument;
615 ✔
129
        $this->xPath = new \DOMXPath($this->domDocument);
615 ✔
130
    }
131

132
    /**
133
     * @throws \UnexpectedValueException
134
     */
135
    protected function getXPath(): \DOMXPath
×
136
    {
137
        if (!$this->xPath instanceof \DOMXPath) {
×
138
            $message = self::class . '::setDomDocument() has not yet been called on ' . static::class;
×
139
            throw new \UnexpectedValueException($message, 1617819086);
×
140
        }
141

142
        return $this->xPath;
×
143
    }
144

145
    /**
146
     * Renders the normalized and processed HTML.
147
     *
148
     * @throws \RuntimeException if there is an internal error with `DOMDocument`
149
     */
150
    public function render(): string
212 ✔
151
    {
152
        return $this->getHtml();
212 ✔
153
    }
154

155
    /**
156
     * Renders the content of the BODY element of the normalized and processed HTML.
157
     *
158
     * @throws \RuntimeException if there is an internal error with `DOMDocument`
159
     */
160
    public function renderBodyContent(): string
12 ✔
161
    {
162
        $bodyNodeHtml = $this->getHtml($this->getBodyElement());
12 ✔
163

164
        return preg_replace('%</?+body(?:\\s[^>]*+)?+>%', '', $bodyNodeHtml);
12 ✔
165
    }
166

167
    /**
168
     * @param ?\DOMNode $node optional parameter to output a subset of the document
169
     *
170
     * @throws \RuntimeException if there is an internal error with `DOMDocument`
171
     */
172
    private function getHtml(?\DOMNode $node = null): string
224 ✔
173
    {
174
        $html = $this->getDomDocument()->saveHTML($node);
224 ✔
175

176
        if (!\is_string($html)) {
224 ✔
NEW
177
            throw new \RuntimeException('`DOMDocument::saveHTML()` failed.', 1773018082);
×
178
        }
179
        return $this->removeSelfClosingTagsClosingTags($html);
224 ✔
180
    }
181

182
    /**
183
     * Eliminates any invalid closing tags for void elements from the given HTML.
184
     */
185
    private function removeSelfClosingTagsClosingTags(string $html): string
224 ✔
186
    {
187
        return preg_replace('%</' . self::PHP_UNRECOGNIZED_VOID_TAGNAME_MATCHER . '>%', '', $html);
224 ✔
188
    }
189

190
    /**
191
     * Returns the HTML element.
192
     *
193
     * This method assumes that there always is an HTML element, throwing an exception otherwise.
194
     *
195
     * @throws \UnexpectedValueException
196
     */
197
    protected function getHtmlElement(): \DOMElement
414 ✔
198
    {
199
        $htmlElement = $this->getDomDocument()->getElementsByTagName('html')->item(0);
414 ✔
200
        if (!$htmlElement instanceof \DOMElement) {
414 ✔
201
            throw new \UnexpectedValueException('There is no HTML element although there should be one.', 1569930853);
×
202
        }
203

204
        return $htmlElement;
414 ✔
205
    }
206

207
    /**
208
     * Returns the BODY element.
209
     *
210
     * This method assumes that there always is a BODY element.
211
     *
212
     * @throws \RuntimeException
213
     */
214
    private function getBodyElement(): \DOMElement
12 ✔
215
    {
216
        $node = $this->getDomDocument()->getElementsByTagName('body')->item(0);
12 ✔
217
        if (!$node instanceof \DOMElement) {
12 ✔
218
            throw new \RuntimeException('There is no body element.', 1617922607);
×
219
        }
220

221
        return $node;
12 ✔
222
    }
223

224
    /**
225
     * Creates a DOM document from the given HTML and stores it in $this->domDocument.
226
     *
227
     * The DOM document will always have a BODY element and a document type.
228
     */
229
    private function createUnifiedDomDocument(string $html): void
611 ✔
230
    {
231
        $this->createRawDomDocument($html);
611 ✔
232
        $this->ensureExistenceOfBodyElement();
611 ✔
233
    }
234

235
    /**
236
     * Creates a DOMDocument instance from the given HTML and stores it in $this->domDocument.
237
     */
238
    private function createRawDomDocument(string $html): void
611 ✔
239
    {
240
        $domDocument = new \DOMDocument();
611 ✔
241
        $domDocument->strictErrorChecking = false;
611 ✔
242
        $domDocument->formatOutput = false;
611 ✔
243
        $libXmlState = \libxml_use_internal_errors(true);
611 ✔
244
        $domDocument->loadHTML($this->prepareHtmlForDomConversion($html), LIBXML_PARSEHUGE);
611 ✔
245
        \libxml_clear_errors();
611 ✔
246
        \libxml_use_internal_errors($libXmlState);
611 ✔
247

248
        $this->setDomDocument($domDocument);
611 ✔
249
    }
250

251
    /**
252
     * Returns the HTML with added document type, Content-Type meta tag, and self-closing slashes, if needed,
253
     * ensuring that the HTML will be good for creating a DOM document from it.
254
     */
255
    private function prepareHtmlForDomConversion(string $html): string
611 ✔
256
    {
257
        $htmlWithSelfClosingSlashes = $this->ensurePhpUnrecognizedSelfClosingTagsAreXml($html);
611 ✔
258
        $htmlWithDocumentType = $this->ensureDocumentType($htmlWithSelfClosingSlashes);
611 ✔
259

260
        return $this->addContentTypeMetaTag($htmlWithDocumentType);
611 ✔
261
    }
262

263
    /**
264
     * Makes sure that the passed HTML has a document type, with lowercase "html".
265
     *
266
     * @return non-empty-string HTML with document type
267
     */
268
    private function ensureDocumentType(string $html): string
611 ✔
269
    {
270
        $hasDocumentType = \stripos($html, '<!DOCTYPE') !== false;
611 ✔
271
        if ($hasDocumentType) {
611 ✔
272
            return $this->normalizeDocumentType($html);
39 ✔
273
        }
274

275
        return self::DEFAULT_DOCUMENT_TYPE . $html;
572 ✔
276
    }
277

278
    /**
279
     * Makes sure the document type in the passed HTML has lowercase `html`.
280
     *
281
     * @param non-empty-string $html
282
     *
283
     * @return non-empty-string HTML with normalized document type
284
     */
285
    private function normalizeDocumentType(string $html): string
39 ✔
286
    {
287
        // Limit to replacing the first occurrence: as an optimization; and in case an example exists as unescaped text.
288
        $result = preg_replace(
39 ✔
289
            '/<!DOCTYPE\\s++html(?=[\\s>])/i',
39 ✔
290
            '<!DOCTYPE html',
39 ✔
291
            $html,
39 ✔
292
            1
39 ✔
293
        );
39 ✔
294
        \assert($result !== '');
39 ✔
295

296
        return $result;
39 ✔
297
    }
298

299
    /**
300
     * Adds a Content-Type meta tag for the charset.
301
     *
302
     * This method also ensures that there is a HEAD element.
303
     *
304
     * @param non-empty-string $html
305
     *
306
     * @return non-empty-string
307
     */
308
    private function addContentTypeMetaTag(string $html): string
611 ✔
309
    {
310
        if ($this->hasContentTypeMetaTagInHead($html)) {
611 ✔
311
            return $html;
374 ✔
312
        }
313

314
        // We are trying to insert the meta tag to the right spot in the DOM.
315
        // If we just prepended it to the HTML, we would lose attributes set to the HTML tag.
316
        $hasHeadTag = preg_match('/<head[\\s>]/i', $html) !== 0;
237 ✔
317
        $hasHtmlTag = \stripos($html, '<html') !== false;
237 ✔
318

319
        if ($hasHeadTag) {
237 ✔
320
            $reworkedHtml = preg_replace(
42 ✔
321
                '/<head(?=[\\s>])([^>]*+)>/i',
42 ✔
322
                '<head$1>' . self::CONTENT_TYPE_META_TAG,
42 ✔
323
                $html
42 ✔
324
            );
42 ✔
325
        } elseif ($hasHtmlTag) {
195 ✔
326
            $reworkedHtml = preg_replace(
83 ✔
327
                '/<html(.*?)>/is',
83 ✔
328
                '<html$1><head>' . self::CONTENT_TYPE_META_TAG . '</head>',
83 ✔
329
                $html
83 ✔
330
            );
83 ✔
331
        } else {
332
            $reworkedHtml = self::CONTENT_TYPE_META_TAG . $html;
112 ✔
333
        }
334
        \assert($reworkedHtml !== '');
237 ✔
335

336
        return $reworkedHtml;
237 ✔
337
    }
338

339
    /**
340
     * Tests whether the given HTML has a valid `Content-Type` metadata element within the `<head>` element.  Due to tag
341
     * omission rules, HTML parsers are expected to end the `<head>` element and start the `<body>` element upon
342
     * encountering a start tag for any element which is permitted only within the `<body>`.
343
     */
344
    private function hasContentTypeMetaTagInHead(string $html): bool
611 ✔
345
    {
346
        preg_match(
611 ✔
347
            '%
611 ✔
348
                (?(DEFINE)
349
                    # the target `http-equiv` attribute match
350
                    (?<target_attribute>
351
                        http-equiv=(["\']?+)Content-Type\\g{-1}
352
                        # must be followed by one of these characters
353
                        [\\s/>]
354
                    )
355
                    # the target `meta` element match without the opening `<`
356
                    (?<target>
357
                        meta(?=\\s)
358
                        # one or other of these
359
                        (?:
360
                            # one or more characters other than `>` or space
361
                            [^>\\s]++
362
                            |
363
                            # space not followed by the target `http-equiv` attribute
364
                            \\s(?!(?&target_attribute))
365
                        )
366
                        # any number of times (including zero)
367
                        *+
368
                        \\s(?&target_attribute)
369
                    )
370
                )
371
                # start of `subject`
372
                ^
373
                # one or other of these
374
                (?:
375
                    # one or more characters other than `<`
376
                    [^<]++
377
                    |
378
                    # `<` not followed by `target`
379
                    <(?!(?&target))
380
                )
381
                # any number of times (including zero)
382
                *+
383
                # followed by the target, not captured
384
                (?=<(?&target))
385
            %isx',
611 ✔
386
            $html,
611 ✔
387
            $matches
611 ✔
388
        );
611 ✔
389
        if (isset($matches[0])) {
611 ✔
390
            $htmlBefore = $matches[0];
396 ✔
391
            try {
392
                $hasContentTypeMetaTagInHead = !$this->hasEndOfHeadElement($htmlBefore);
396 ✔
393
            } catch (\RuntimeException $exception) {
×
394
                // If something unexpected occurs, assume the `Content-Type` that was found is valid.
395
                \trigger_error($exception->getMessage());
×
396
                $hasContentTypeMetaTagInHead = true;
×
397
            }
398
        } else {
399
            $hasContentTypeMetaTagInHead = false;
215 ✔
400
        }
401

402
        return $hasContentTypeMetaTagInHead;
611 ✔
403
    }
404

405
    /**
406
     * Tests whether the `<head>` element ends within the given HTML.  Due to tag omission rules, HTML parsers are
407
     * expected to end the `<head>` element and start the `<body>` element upon encountering a start tag for any element
408
     * which is permitted only within the `<body>`.
409
     *
410
     * @throws \RuntimeException
411
     */
412
    private function hasEndOfHeadElement(string $html): bool
396 ✔
413
    {
414
        if (preg_match('%<(?!' . self::TAGNAME_ALLOWED_BEFORE_BODY_MATCHER . '[\\s/>])\\w|</head>%i', $html) !== 0) {
396 ✔
415
            // An exception to the implicit end of the `<head>` is any content within a `<template>` element, as well in
416
            // comments.  As an optimization, this is only checked for if a potential `<head>` end tag is found.
417
            $htmlWithoutCommentsOrTemplates = $this->removeHtmlTemplateElements($this->removeHtmlComments($html));
70 ✔
418
            $hasEndOfHeadElement = $htmlWithoutCommentsOrTemplates === $html
70 ✔
419
                || $this->hasEndOfHeadElement($htmlWithoutCommentsOrTemplates);
70 ✔
420
        } else {
421
            $hasEndOfHeadElement = false;
374 ✔
422
        }
423

424
        return $hasEndOfHeadElement;
396 ✔
425
    }
426

427
    /**
428
     * Removes comments from the given HTML, including any which are unterminated, for which the remainder of the string
429
     * is removed.
430
     */
431
    private function removeHtmlComments(string $html): string
70 ✔
432
    {
433
        return preg_replace(self::HTML_COMMENT_PATTERN, '', $html);
70 ✔
434
    }
435

436
    /**
437
     * Removes `<template>` elements from the given HTML, including any without an end tag, for which the remainder of
438
     * the string is removed.
439
     */
440
    private function removeHtmlTemplateElements(string $html): string
70 ✔
441
    {
442
        return preg_replace(self::HTML_TEMPLATE_ELEMENT_PATTERN, '', $html);
70 ✔
443
    }
444

445
    /**
446
     * Makes sure that any self-closing tags not recognized as such by PHP's DOMDocument implementation have a
447
     * self-closing slash.
448
     */
449
    private function ensurePhpUnrecognizedSelfClosingTagsAreXml(string $html): string
611 ✔
450
    {
451
        return preg_replace(
611 ✔
452
            '%<' . self::PHP_UNRECOGNIZED_VOID_TAGNAME_MATCHER . '\\b[^>]*+(?<!/)(?=>)%',
611 ✔
453
            '$0/',
611 ✔
454
            $html
611 ✔
455
        );
611 ✔
456
    }
457

458
    /**
459
     * Checks that $this->domDocument has a BODY element and adds it if it is missing.
460
     *
461
     * @throws \UnexpectedValueException
462
     */
463
    private function ensureExistenceOfBodyElement(): void
611 ✔
464
    {
465
        if ($this->getDomDocument()->getElementsByTagName('body')->item(0) instanceof \DOMElement) {
611 ✔
466
            return;
198 ✔
467
        }
468

469
        $this->getHtmlElement()->appendChild($this->getDomDocument()->createElement('body'));
413 ✔
470
    }
471
}
STATUS · Troubleshooting · Open an Issue · Sales · Support · CAREERS · ENTERPRISE · START FREE TRIAL · SCHEDULE DEMO
ANNOUNCEMENTS · TWITTER · TOS & SLA · Supported CI Services · What's a CI service? · Automated Testing

© 2026 Coveralls, Inc