PluginProbe
BeyondWords – AI audio for publishers / 7.2.0
BeyondWords – AI audio for publishers v7.2.0
7.2.0 7.1.0 trunk 4.0.0 4.0.1 4.0.2 4.0.3 4.0.4 4.0.5 4.0.6 4.1.0 4.1.1 4.1.2 4.2.0 4.2.1 4.2.2 4.2.3 4.2.4 4.3.0 4.4.0 4.5.0 4.5.1 4.6.0 4.6.1 4.6.2 All 44 releases
speechkit / vendor / symfony / dom-crawler / Crawler.php

Crawler.php in BeyondWords – AI audio for publishers 7.2.0, at vendor/symfony/dom-crawler/Crawler.php

1,327 lines 39.0 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2
3 /*
4 * This file is part of the Symfony package.
5 *
6 * (c) Fabien Potencier <[email protected]>
7 *
8 * For the full copyright and license information, please view the LICENSE
9 * file that was distributed with this source code.
10 */
11
12 namespace Symfony\Component\DomCrawler;
13
14 use Masterminds\HTML5;
15 use Symfony\Component\CssSelector\CssSelectorConverter;
16
17 /**
18 * Crawler eases navigation of a list of \DOMNode objects.
19 *
20 * @author Fabien Potencier <[email protected]>
21 *
22 * @implements \IteratorAggregate<int, \DOMNode>
23 */
24 class Crawler implements \Countable, \IteratorAggregate
25 {
26 /**
27 * @var string|null
28 */
29 protected $uri;
30
31 /**
32 * The default namespace prefix to be used with XPath and CSS expressions.
33 *
34 * @var string
35 */
36 private $defaultNamespacePrefix = 'default';
37
38 /**
39 * A map of manually registered namespaces.
40 *
41 * @var array<string, string>
42 */
43 private $namespaces = [];
44
45 /**
46 * A map of cached namespaces.
47 *
48 * @var \ArrayObject
49 */
50 private $cachedNamespaces;
51
52 /**
53 * The base href value.
54 *
55 * @var string|null
56 */
57 private $baseHref;
58
59 /**
60 * @var \DOMDocument|null
61 */
62 private $document;
63
64 /**
65 * @var list<\DOMNode>
66 */
67 private $nodes = [];
68
69 /**
70 * Whether the Crawler contains HTML or XML content (used when converting CSS to XPath).
71 *
72 * @var bool
73 */
74 private $isHtml = true;
75
76 /**
77 * @var HTML5|null
78 */
79 private $html5Parser;
80
81 /**
82 * @param \DOMNodeList|\DOMNode|\DOMNode[]|string|null $node A Node to use as the base for the crawling
83 */
84 public function __construct($node = null, ?string $uri = null, ?string $baseHref = null)
85 {
86 $this->uri = $uri;
87 $this->baseHref = $baseHref ?: $uri;
88 $this->html5Parser = class_exists(HTML5::class) ? new HTML5(['disable_html_ns' => true]) : null;
89 $this->cachedNamespaces = new \ArrayObject();
90
91 $this->add($node);
92 }
93
94 /**
95 * Returns the current URI.
96 *
97 * @return string|null
98 */
99 public function getUri()
100 {
101 return $this->uri;
102 }
103
104 /**
105 * Returns base href.
106 *
107 * @return string|null
108 */
109 public function getBaseHref()
110 {
111 return $this->baseHref;
112 }
113
114 /**
115 * Removes all the nodes.
116 */
117 public function clear()
118 {
119 $this->nodes = [];
120 $this->document = null;
121 $this->cachedNamespaces = new \ArrayObject();
122 }
123
124 /**
125 * Adds a node to the current list of nodes.
126 *
127 * This method uses the appropriate specialized add*() method based
128 * on the type of the argument.
129 *
130 * @param \DOMNodeList|\DOMNode|\DOMNode[]|string|null $node A node
131 *
132 * @throws \InvalidArgumentException when node is not the expected type
133 */
134 public function add($node)
135 {
136 if ($node instanceof \DOMNodeList) {
137 $this->addNodeList($node);
138 } elseif ($node instanceof \DOMNode) {
139 $this->addNode($node);
140 } elseif (\is_array($node)) {
141 $this->addNodes($node);
142 } elseif (\is_string($node)) {
143 $this->addContent($node);
144 } elseif (null !== $node) {
145 throw new \InvalidArgumentException(sprintf('Expecting a DOMNodeList or DOMNode instance, an array, a string, or null, but got "%s".', get_debug_type($node)));
146 }
147 }
148
149 /**
150 * Adds HTML/XML content.
151 *
152 * If the charset is not set via the content type, it is assumed to be UTF-8,
153 * or ISO-8859-1 as a fallback, which is the default charset defined by the
154 * HTTP 1.1 specification.
155 */
156 public function addContent(string $content, ?string $type = null)
157 {
158 if (empty($type)) {
159 $type = str_starts_with($content, '<?xml') ? 'application/xml' : 'text/html';
160 }
161
162 // DOM only for HTML/XML content
163 if (!preg_match('/(x|ht)ml/i', $type, $xmlMatches)) {
164 return;
165 }
166
167 $charset = preg_match('//u', $content) ? 'UTF-8' : 'ISO-8859-1';
168
169 // http://www.w3.org/TR/encoding/#encodings
170 // http://www.w3.org/TR/REC-xml/#NT-EncName
171 $content = preg_replace_callback('/(charset *= *["\']?)([a-zA-Z\-0-9_:.]+)/i', function ($m) use (&$charset) {
172 if ('charset=' === $this->convertToHtmlEntities('charset=', $m[2])) {
173 $charset = $m[2];
174 }
175
176 return $m[1].$charset;
177 }, $content, 1);
178
179 if ('x' === $xmlMatches[1]) {
180 $this->addXmlContent($content, $charset);
181 } else {
182 $this->addHtmlContent($content, $charset);
183 }
184 }
185
186 /**
187 * Adds an HTML content to the list of nodes.
188 *
189 * The libxml errors are disabled when the content is parsed.
190 *
191 * If you want to get parsing errors, be sure to enable
192 * internal errors via libxml_use_internal_errors(true)
193 * and then, get the errors via libxml_get_errors(). Be
194 * sure to clear errors with libxml_clear_errors() afterward.
195 */
196 public function addHtmlContent(string $content, string $charset = 'UTF-8')
197 {
198 $dom = $this->parseHtmlString($content, $charset);
199 $this->addDocument($dom);
200
201 $base = $this->filterRelativeXPath('descendant-or-self::base')->extract(['href']);
202
203 $baseHref = current($base);
204 if (\count($base) && !empty($baseHref)) {
205 if ($this->baseHref) {
206 $linkNode = $dom->createElement('a');
207 $linkNode->setAttribute('href', $baseHref);
208 $link = new Link($linkNode, $this->baseHref);
209 $this->baseHref = $link->getUri();
210 } else {
211 $this->baseHref = $baseHref;
212 }
213 }
214 }
215
216 /**
217 * Adds an XML content to the list of nodes.
218 *
219 * The libxml errors are disabled when the content is parsed.
220 *
221 * If you want to get parsing errors, be sure to enable
222 * internal errors via libxml_use_internal_errors(true)
223 * and then, get the errors via libxml_get_errors(). Be
224 * sure to clear errors with libxml_clear_errors() afterward.
225 *
226 * @param int $options Bitwise OR of the libxml option constants
227 * LIBXML_PARSEHUGE is dangerous, see
228 * http://symfony.com/blog/security-release-symfony-2-0-17-released
229 */
230 public function addXmlContent(string $content, string $charset = 'UTF-8', int $options = \LIBXML_NONET)
231 {
232 // remove the default namespace if it's the only namespace to make XPath expressions simpler
233 if (!preg_match('/xmlns:/', $content)) {
234 $content = str_replace('xmlns', 'ns', $content);
235 }
236
237 $internalErrors = libxml_use_internal_errors(true);
238 if (\LIBXML_VERSION < 20900) {
239 $disableEntities = libxml_disable_entity_loader(true);
240 }
241
242 $dom = new \DOMDocument('1.0', $charset);
243
244 if ('' !== trim($content)) {
245 @$dom->loadXML($content, $options);
246 }
247
248 libxml_use_internal_errors($internalErrors);
249 if (\LIBXML_VERSION < 20900) {
250 libxml_disable_entity_loader($disableEntities);
251 }
252
253 $this->addDocument($dom);
254
255 $this->isHtml = false;
256 }
257
258 /**
259 * Adds a \DOMDocument to the list of nodes.
260 *
261 * @param \DOMDocument $dom A \DOMDocument instance
262 */
263 public function addDocument(\DOMDocument $dom)
264 {
265 if ($dom->documentElement) {
266 $this->addNode($dom->documentElement);
267 }
268 }
269
270 /**
271 * Adds a \DOMNodeList to the list of nodes.
272 *
273 * @param \DOMNodeList $nodes A \DOMNodeList instance
274 */
275 public function addNodeList(\DOMNodeList $nodes)
276 {
277 foreach ($nodes as $node) {
278 if ($node instanceof \DOMNode) {
279 $this->addNode($node);
280 }
281 }
282 }
283
284 /**
285 * Adds an array of \DOMNode instances to the list of nodes.
286 *
287 * @param \DOMNode[] $nodes An array of \DOMNode instances
288 */
289 public function addNodes(array $nodes)
290 {
291 foreach ($nodes as $node) {
292 $this->add($node);
293 }
294 }
295
296 /**
297 * Adds a \DOMNode instance to the list of nodes.
298 *
299 * @param \DOMNode $node A \DOMNode instance
300 */
301 public function addNode(\DOMNode $node)
302 {
303 if ($node instanceof \DOMDocument) {
304 $node = $node->documentElement;
305 }
306
307 if (null !== $this->document && $this->document !== $node->ownerDocument) {
308 throw new \InvalidArgumentException('Attaching DOM nodes from multiple documents in the same crawler is forbidden.');
309 }
310
311 if (null === $this->document) {
312 $this->document = $node->ownerDocument;
313 }
314
315 // Don't add duplicate nodes in the Crawler
316 if (\in_array($node, $this->nodes, true)) {
317 return;
318 }
319
320 $this->nodes[] = $node;
321 }
322
323 /**
324 * Returns a node given its position in the node list.
325 *
326 * @return static
327 */
328 public function eq(int $position)
329 {
330 if (isset($this->nodes[$position])) {
331 return $this->createSubCrawler($this->nodes[$position]);
332 }
333
334 return $this->createSubCrawler(null);
335 }
336
337 /**
338 * Calls an anonymous function on each node of the list.
339 *
340 * The anonymous function receives the position and the node wrapped
341 * in a Crawler instance as arguments.
342 *
343 * Example:
344 *
345 * $crawler->filter('h1')->each(function ($node, $i) {
346 * return $node->text();
347 * });
348 *
349 * @param \Closure $closure An anonymous function
350 *
351 * @return array An array of values returned by the anonymous function
352 */
353 public function each(\Closure $closure)
354 {
355 $data = [];
356 foreach ($this->nodes as $i => $node) {
357 $data[] = $closure($this->createSubCrawler($node), $i);
358 }
359
360 return $data;
361 }
362
363 /**
364 * Slices the list of nodes by $offset and $length.
365 *
366 * @return static
367 */
368 public function slice(int $offset = 0, ?int $length = null)
369 {
370 return $this->createSubCrawler(\array_slice($this->nodes, $offset, $length));
371 }
372
373 /**
374 * Reduces the list of nodes by calling an anonymous function.
375 *
376 * To remove a node from the list, the anonymous function must return false.
377 *
378 * @param \Closure $closure An anonymous function
379 *
380 * @return static
381 */
382 public function reduce(\Closure $closure)
383 {
384 $nodes = [];
385 foreach ($this->nodes as $i => $node) {
386 if (false !== $closure($this->createSubCrawler($node), $i)) {
387 $nodes[] = $node;
388 }
389 }
390
391 return $this->createSubCrawler($nodes);
392 }
393
394 /**
395 * Returns the first node of the current selection.
396 *
397 * @return static
398 */
399 public function first()
400 {
401 return $this->eq(0);
402 }
403
404 /**
405 * Returns the last node of the current selection.
406 *
407 * @return static
408 */
409 public function last()
410 {
411 return $this->eq(\count($this->nodes) - 1);
412 }
413
414 /**
415 * Returns the siblings nodes of the current selection.
416 *
417 * @return static
418 *
419 * @throws \InvalidArgumentException When current node is empty
420 */
421 public function siblings()
422 {
423 if (!$this->nodes) {
424 throw new \InvalidArgumentException('The current node list is empty.');
425 }
426
427 return $this->createSubCrawler($this->sibling($this->getNode(0)->parentNode->firstChild));
428 }
429
430 public function matches(string $selector): bool
431 {
432 if (!$this->nodes) {
433 return false;
434 }
435
436 $converter = $this->createCssSelectorConverter();
437 $xpath = $converter->toXPath($selector, 'self::');
438
439 return 0 !== $this->filterRelativeXPath($xpath)->count();
440 }
441
442 /**
443 * Return first parents (heading toward the document root) of the Element that matches the provided selector.
444 *
445 * @see https://developer.mozilla.org/en-US/docs/Web/API/Element/closest#Polyfill
446 *
447 * @throws \InvalidArgumentException When current node is empty
448 */
449 public function closest(string $selector): ?self
450 {
451 if (!$this->nodes) {
452 throw new \InvalidArgumentException('The current node list is empty.');
453 }
454
455 $domNode = $this->getNode(0);
456
457 while (\XML_ELEMENT_NODE === $domNode->nodeType) {
458 $node = $this->createSubCrawler($domNode);
459 if ($node->matches($selector)) {
460 return $node;
461 }
462
463 $domNode = $node->getNode(0)->parentNode;
464 }
465
466 return null;
467 }
468
469 /**
470 * Returns the next siblings nodes of the current selection.
471 *
472 * @return static
473 *
474 * @throws \InvalidArgumentException When current node is empty
475 */
476 public function nextAll()
477 {
478 if (!$this->nodes) {
479 throw new \InvalidArgumentException('The current node list is empty.');
480 }
481
482 return $this->createSubCrawler($this->sibling($this->getNode(0)));
483 }
484
485 /**
486 * Returns the previous sibling nodes of the current selection.
487 *
488 * @return static
489 *
490 * @throws \InvalidArgumentException
491 */
492 public function previousAll()
493 {
494 if (!$this->nodes) {
495 throw new \InvalidArgumentException('The current node list is empty.');
496 }
497
498 return $this->createSubCrawler($this->sibling($this->getNode(0), 'previousSibling'));
499 }
500
501 /**
502 * Returns the parent nodes of the current selection.
503 *
504 * @return static
505 *
506 * @throws \InvalidArgumentException When current node is empty
507 */
508 public function parents()
509 {
510 trigger_deprecation('symfony/dom-crawler', '5.3', 'The %s() method is deprecated, use ancestors() instead.', __METHOD__);
511
512 return $this->ancestors();
513 }
514
515 /**
516 * Returns the ancestors of the current selection.
517 *
518 * @return static
519 *
520 * @throws \InvalidArgumentException When the current node is empty
521 */
522 public function ancestors()
523 {
524 if (!$this->nodes) {
525 throw new \InvalidArgumentException('The current node list is empty.');
526 }
527
528 $node = $this->getNode(0);
529 $nodes = [];
530
531 while ($node = $node->parentNode) {
532 if (\XML_ELEMENT_NODE === $node->nodeType) {
533 $nodes[] = $node;
534 }
535 }
536
537 return $this->createSubCrawler($nodes);
538 }
539
540 /**
541 * Returns the children nodes of the current selection.
542 *
543 * @return static
544 *
545 * @throws \InvalidArgumentException When current node is empty
546 * @throws \RuntimeException If the CssSelector Component is not available and $selector is provided
547 */
548 public function children(?string $selector = null)
549 {
550 if (!$this->nodes) {
551 throw new \InvalidArgumentException('The current node list is empty.');
552 }
553
554 if (null !== $selector) {
555 $converter = $this->createCssSelectorConverter();
556 $xpath = $converter->toXPath($selector, 'child::');
557
558 return $this->filterRelativeXPath($xpath);
559 }
560
561 $node = $this->getNode(0)->firstChild;
562
563 return $this->createSubCrawler($node ? $this->sibling($node) : []);
564 }
565
566 /**
567 * Returns the attribute value of the first node of the list.
568 *
569 * @return string|null
570 *
571 * @throws \InvalidArgumentException When current node is empty
572 */
573 public function attr(string $attribute)
574 {
575 if (!$this->nodes) {
576 throw new \InvalidArgumentException('The current node list is empty.');
577 }
578
579 $node = $this->getNode(0);
580
581 return $node->hasAttribute($attribute) ? $node->getAttribute($attribute) : null;
582 }
583
584 /**
585 * Returns the node name of the first node of the list.
586 *
587 * @return string
588 *
589 * @throws \InvalidArgumentException When current node is empty
590 */
591 public function nodeName()
592 {
593 if (!$this->nodes) {
594 throw new \InvalidArgumentException('The current node list is empty.');
595 }
596
597 return $this->getNode(0)->nodeName;
598 }
599
600 /**
601 * Returns the text of the first node of the list.
602 *
603 * Pass true as the second argument to normalize whitespaces.
604 *
605 * @param string|null $default When not null: the value to return when the current node is empty
606 * @param bool $normalizeWhitespace Whether whitespaces should be trimmed and normalized to single spaces
607 *
608 * @return string
609 *
610 * @throws \InvalidArgumentException When current node is empty
611 */
612 public function text(?string $default = null, bool $normalizeWhitespace = true)
613 {
614 if (!$this->nodes) {
615 if (null !== $default) {
616 return $default;
617 }
618
619 throw new \InvalidArgumentException('The current node list is empty.');
620 }
621
622 $text = $this->getNode(0)->nodeValue;
623
624 if ($normalizeWhitespace) {
625 return trim(preg_replace("/(?:[ \n\r\t\x0C]{2,}+|[\n\r\t\x0C])/", ' ', $text), " \n\r\t\x0C");
626 }
627
628 return $text;
629 }
630
631 /**
632 * Returns only the inner text that is the direct descendent of the current node, excluding any child nodes.
633 */
634 public function innerText(): string
635 {
636 return $this->filterXPath('.//text()')->text();
637 }
638
639 /**
640 * Returns the first node of the list as HTML.
641 *
642 * @param string|null $default When not null: the value to return when the current node is empty
643 *
644 * @return string
645 *
646 * @throws \InvalidArgumentException When current node is empty
647 */
648 public function html(?string $default = null)
649 {
650 if (!$this->nodes) {
651 if (null !== $default) {
652 return $default;
653 }
654
655 throw new \InvalidArgumentException('The current node list is empty.');
656 }
657
658 $node = $this->getNode(0);
659 $owner = $node->ownerDocument;
660
661 if (null !== $this->html5Parser && '<!DOCTYPE html>' === $owner->saveXML($owner->childNodes[0])) {
662 $owner = $this->html5Parser;
663 }
664
665 $html = '';
666 foreach ($node->childNodes as $child) {
667 $html .= $owner->saveHTML($child);
668 }
669
670 return $html;
671 }
672
673 public function outerHtml(): string
674 {
675 if (!\count($this)) {
676 throw new \InvalidArgumentException('The current node list is empty.');
677 }
678
679 $node = $this->getNode(0);
680 $owner = $node->ownerDocument;
681
682 if (null !== $this->html5Parser && '<!DOCTYPE html>' === $owner->saveXML($owner->childNodes[0])) {
683 $owner = $this->html5Parser;
684 }
685
686 return $owner->saveHTML($node);
687 }
688
689 /**
690 * Evaluates an XPath expression.
691 *
692 * Since an XPath expression might evaluate to either a simple type or a \DOMNodeList,
693 * this method will return either an array of simple types or a new Crawler instance.
694 *
695 * @return array|Crawler
696 */
697 public function evaluate(string $xpath)
698 {
699 if (null === $this->document) {
700 throw new \LogicException('Cannot evaluate the expression on an uninitialized crawler.');
701 }
702
703 $data = [];
704 $domxpath = $this->createDOMXPath($this->document, $this->findNamespacePrefixes($xpath));
705
706 foreach ($this->nodes as $node) {
707 $data[] = $domxpath->evaluate($xpath, $node);
708 }
709
710 if (isset($data[0]) && $data[0] instanceof \DOMNodeList) {
711 return $this->createSubCrawler($data);
712 }
713
714 return $data;
715 }
716
717 /**
718 * Extracts information from the list of nodes.
719 *
720 * You can extract attributes or/and the node value (_text).
721 *
722 * Example:
723 *
724 * $crawler->filter('h1 a')->extract(['_text', 'href']);
725 *
726 * @return array
727 */
728 public function extract(array $attributes)
729 {
730 $count = \count($attributes);
731
732 $data = [];
733 foreach ($this->nodes as $node) {
734 $elements = [];
735 foreach ($attributes as $attribute) {
736 if ('_text' === $attribute) {
737 $elements[] = $node->nodeValue;
738 } elseif ('_name' === $attribute) {
739 $elements[] = $node->nodeName;
740 } else {
741 $elements[] = $node->getAttribute($attribute);
742 }
743 }
744
745 $data[] = 1 === $count ? $elements[0] : $elements;
746 }
747
748 return $data;
749 }
750
751 /**
752 * Filters the list of nodes with an XPath expression.
753 *
754 * The XPath expression is evaluated in the context of the crawler, which
755 * is considered as a fake parent of the elements inside it.
756 * This means that a child selector "div" or "./div" will match only
757 * the div elements of the current crawler, not their children.
758 *
759 * @return static
760 */
761 public function filterXPath(string $xpath)
762 {
763 $xpath = $this->relativize($xpath);
764
765 // If we dropped all expressions in the XPath while preparing it, there would be no match
766 if ('' === $xpath) {
767 return $this->createSubCrawler(null);
768 }
769
770 return $this->filterRelativeXPath($xpath);
771 }
772
773 /**
774 * Filters the list of nodes with a CSS selector.
775 *
776 * This method only works if you have installed the CssSelector Symfony Component.
777 *
778 * @return static
779 *
780 * @throws \LogicException if the CssSelector Component is not available
781 */
782 public function filter(string $selector)
783 {
784 $converter = $this->createCssSelectorConverter();
785
786 // The CssSelector already prefixes the selector with descendant-or-self::
787 return $this->filterRelativeXPath($converter->toXPath($selector));
788 }
789
790 /**
791 * Selects links by name or alt value for clickable images.
792 *
793 * @return static
794 */
795 public function selectLink(string $value)
796 {
797 return $this->filterRelativeXPath(
798 sprintf('descendant-or-self::a[contains(concat(\' \', normalize-space(string(.)), \' \'), %1$s) or ./img[contains(concat(\' \', normalize-space(string(@alt)), \' \'), %1$s)]]', static::xpathLiteral(' '.$value.' '))
799 );
800 }
801
802 /**
803 * Selects images by alt value.
804 *
805 * @return static
806 */
807 public function selectImage(string $value)
808 {
809 $xpath = sprintf('descendant-or-self::img[contains(normalize-space(string(@alt)), %s)]', static::xpathLiteral($value));
810
811 return $this->filterRelativeXPath($xpath);
812 }
813
814 /**
815 * Selects a button by name or alt value for images.
816 *
817 * @return static
818 */
819 public function selectButton(string $value)
820 {
821 return $this->filterRelativeXPath(
822 sprintf('descendant-or-self::input[((contains(%1$s, "submit") or contains(%1$s, "button")) and contains(concat(\' \', normalize-space(string(@value)), \' \'), %2$s)) or (contains(%1$s, "image") and contains(concat(\' \', normalize-space(string(@alt)), \' \'), %2$s)) or @id=%3$s or @name=%3$s] | descendant-or-self::button[contains(concat(\' \', normalize-space(string(.)), \' \'), %2$s) or @id=%3$s or @name=%3$s]', 'translate(@type, "ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz")', static::xpathLiteral(' '.$value.' '), static::xpathLiteral($value))
823 );
824 }
825
826 /**
827 * Returns a Link object for the first node in the list.
828 *
829 * @return Link
830 *
831 * @throws \InvalidArgumentException If the current node list is empty or the selected node is not instance of DOMElement
832 */
833 public function link(string $method = 'get')
834 {
835 if (!$this->nodes) {
836 throw new \InvalidArgumentException('The current node list is empty.');
837 }
838
839 $node = $this->getNode(0);
840
841 if (!$node instanceof \DOMElement) {
842 throw new \InvalidArgumentException(sprintf('The selected node should be instance of DOMElement, got "%s".', get_debug_type($node)));
843 }
844
845 return new Link($node, $this->baseHref, $method);
846 }
847
848 /**
849 * Returns an array of Link objects for the nodes in the list.
850 *
851 * @return Link[]
852 *
853 * @throws \InvalidArgumentException If the current node list contains non-DOMElement instances
854 */
855 public function links()
856 {
857 $links = [];
858 foreach ($this->nodes as $node) {
859 if (!$node instanceof \DOMElement) {
860 throw new \InvalidArgumentException(sprintf('The current node list should contain only DOMElement instances, "%s" found.', get_debug_type($node)));
861 }
862
863 $links[] = new Link($node, $this->baseHref, 'get');
864 }
865
866 return $links;
867 }
868
869 /**
870 * Returns an Image object for the first node in the list.
871 *
872 * @return Image
873 *
874 * @throws \InvalidArgumentException If the current node list is empty
875 */
876 public function image()
877 {
878 if (!\count($this)) {
879 throw new \InvalidArgumentException('The current node list is empty.');
880 }
881
882 $node = $this->getNode(0);
883
884 if (!$node instanceof \DOMElement) {
885 throw new \InvalidArgumentException(sprintf('The selected node should be instance of DOMElement, got "%s".', get_debug_type($node)));
886 }
887
888 return new Image($node, $this->baseHref);
889 }
890
891 /**
892 * Returns an array of Image objects for the nodes in the list.
893 *
894 * @return Image[]
895 */
896 public function images()
897 {
898 $images = [];
899 foreach ($this as $node) {
900 if (!$node instanceof \DOMElement) {
901 throw new \InvalidArgumentException(sprintf('The current node list should contain only DOMElement instances, "%s" found.', get_debug_type($node)));
902 }
903
904 $images[] = new Image($node, $this->baseHref);
905 }
906
907 return $images;
908 }
909
910 /**
911 * Returns a Form object for the first node in the list.
912 *
913 * @return Form
914 *
915 * @throws \InvalidArgumentException If the current node list is empty or the selected node is not instance of DOMElement
916 */
917 public function form(?array $values = null, ?string $method = null)
918 {
919 if (!$this->nodes) {
920 throw new \InvalidArgumentException('The current node list is empty.');
921 }
922
923 $node = $this->getNode(0);
924
925 if (!$node instanceof \DOMElement) {
926 throw new \InvalidArgumentException(sprintf('The selected node should be instance of DOMElement, got "%s".', get_debug_type($node)));
927 }
928
929 $form = new Form($node, $this->uri, $method, $this->baseHref);
930
931 if (null !== $values) {
932 $form->setValues($values);
933 }
934
935 return $form;
936 }
937
938 /**
939 * Overloads a default namespace prefix to be used with XPath and CSS expressions.
940 */
941 public function setDefaultNamespacePrefix(string $prefix)
942 {
943 $this->defaultNamespacePrefix = $prefix;
944 }
945
946 public function registerNamespace(string $prefix, string $namespace)
947 {
948 $this->namespaces[$prefix] = $namespace;
949 }
950
951 /**
952 * Converts string for XPath expressions.
953 *
954 * Escaped characters are: quotes (") and apostrophe (').
955 *
956 * Examples:
957 *
958 * echo Crawler::xpathLiteral('foo " bar');
959 * //prints 'foo " bar'
960 *
961 * echo Crawler::xpathLiteral("foo ' bar");
962 * //prints "foo ' bar"
963 *
964 * echo Crawler::xpathLiteral('a\'b"c');
965 * //prints concat('a', "'", 'b"c')
966 *
967 * @return string
968 */
969 public static function xpathLiteral(string $s)
970 {
971 if (!str_contains($s, "'")) {
972 return sprintf("'%s'", $s);
973 }
974
975 if (!str_contains($s, '"')) {
976 return sprintf('"%s"', $s);
977 }
978
979 $string = $s;
980 $parts = [];
981 while (true) {
982 if (false !== $pos = strpos($string, "'")) {
983 $parts[] = sprintf("'%s'", substr($string, 0, $pos));
984 $parts[] = "\"'\"";
985 $string = substr($string, $pos + 1);
986 } else {
987 $parts[] = "'$string'";
988 break;
989 }
990 }
991
992 return sprintf('concat(%s)', implode(', ', $parts));
993 }
994
995 /**
996 * Filters the list of nodes with an XPath expression.
997 *
998 * The XPath expression should already be processed to apply it in the context of each node.
999 *
1000 * @return static
1001 */
1002 private function filterRelativeXPath(string $xpath): object
1003 {
1004 $crawler = $this->createSubCrawler(null);
1005 if (null === $this->document) {
1006 return $crawler;
1007 }
1008
1009 $domxpath = $this->createDOMXPath($this->document, $this->findNamespacePrefixes($xpath));
1010
1011 foreach ($this->nodes as $node) {
1012 $crawler->add($domxpath->query($xpath, $node));
1013 }
1014
1015 return $crawler;
1016 }
1017
1018 /**
1019 * Make the XPath relative to the current context.
1020 *
1021 * The returned XPath will match elements matching the XPath inside the current crawler
1022 * when running in the context of a node of the crawler.
1023 */
1024 private function relativize(string $xpath): string
1025 {
1026 $expressions = [];
1027
1028 // An expression which will never match to replace expressions which cannot match in the crawler
1029 // We cannot drop
1030 $nonMatchingExpression = 'a[name() = "b"]';
1031
1032 $xpathLen = \strlen($xpath);
1033 $openedBrackets = 0;
1034 $startPosition = strspn($xpath, " \t\n\r\0\x0B");
1035
1036 for ($i = $startPosition; $i <= $xpathLen; ++$i) {
1037 $i += strcspn($xpath, '"\'[]|', $i);
1038
1039 if ($i < $xpathLen) {
1040 switch ($xpath[$i]) {
1041 case '"':
1042 case "'":
1043 if (false === $i = strpos($xpath, $xpath[$i], $i + 1)) {
1044 return $xpath; // The XPath expression is invalid
1045 }
1046 continue 2;
1047 case '[':
1048 ++$openedBrackets;
1049 continue 2;
1050 case ']':
1051 --$openedBrackets;
1052 continue 2;
1053 }
1054 }
1055 if ($openedBrackets) {
1056 continue;
1057 }
1058
1059 if ($startPosition < $xpathLen && '(' === $xpath[$startPosition]) {
1060 // If the union is inside some braces, we need to preserve the opening braces and apply
1061 // the change only inside it.
1062 $j = 1 + strspn($xpath, "( \t\n\r\0\x0B", $startPosition + 1);
1063 $parenthesis = substr($xpath, $startPosition, $j);
1064 $startPosition += $j;
1065 } else {
1066 $parenthesis = '';
1067 }
1068 $expression = rtrim(substr($xpath, $startPosition, $i - $startPosition));
1069
1070 if (str_starts_with($expression, 'self::*/')) {
1071 $expression = './'.substr($expression, 8);
1072 }
1073
1074 // add prefix before absolute element selector
1075 if ('' === $expression) {
1076 $expression = $nonMatchingExpression;
1077 } elseif (str_starts_with($expression, '//')) {
1078 $expression = 'descendant-or-self::'.substr($expression, 2);
1079 } elseif (str_starts_with($expression, './/')) {
1080 $expression = 'descendant-or-self::'.substr($expression, 3);
1081 } elseif (str_starts_with($expression, './')) {
1082 $expression = 'self::'.substr($expression, 2);
1083 } elseif (str_starts_with($expression, 'child::')) {
1084 $expression = 'self::'.substr($expression, 7);
1085 } elseif ('/' === $expression[0] || '.' === $expression[0] || str_starts_with($expression, 'self::')) {
1086 $expression = $nonMatchingExpression;
1087 } elseif (str_starts_with($expression, 'descendant::')) {
1088 $expression = 'descendant-or-self::'.substr($expression, 12);
1089 } elseif (preg_match('/^(ancestor|ancestor-or-self|attribute|following|following-sibling|namespace|parent|preceding|preceding-sibling)::/', $expression)) {
1090 // the fake root has no parent, preceding or following nodes and also no attributes (even no namespace attributes)
1091 $expression = $nonMatchingExpression;
1092 } elseif (!str_starts_with($expression, 'descendant-or-self::')) {
1093 $expression = 'self::'.$expression;
1094 }
1095 $expressions[] = $parenthesis.$expression;
1096
1097 if ($i === $xpathLen) {
1098 return implode(' | ', $expressions);
1099 }
1100
1101 $i += strspn($xpath, " \t\n\r\0\x0B", $i + 1);
1102 $startPosition = $i + 1;
1103 }
1104
1105 return $xpath; // The XPath expression is invalid
1106 }
1107
1108 /**
1109 * @return \DOMNode|null
1110 */
1111 public function getNode(int $position)
1112 {
1113 return $this->nodes[$position] ?? null;
1114 }
1115
1116 /**
1117 * @return int
1118 */
1119 #[\ReturnTypeWillChange]
1120 public function count()
1121 {
1122 return \count($this->nodes);
1123 }
1124
1125 /**
1126 * @return \ArrayIterator<int, \DOMNode>
1127 */
1128 #[\ReturnTypeWillChange]
1129 public function getIterator()
1130 {
1131 return new \ArrayIterator($this->nodes);
1132 }
1133
1134 /**
1135 * @return array
1136 */
1137 protected function sibling(\DOMNode $node, string $siblingDir = 'nextSibling')
1138 {
1139 $nodes = [];
1140
1141 $currentNode = $this->getNode(0);
1142 do {
1143 if ($node !== $currentNode && \XML_ELEMENT_NODE === $node->nodeType) {
1144 $nodes[] = $node;
1145 }
1146 } while ($node = $node->$siblingDir);
1147
1148 return $nodes;
1149 }
1150
1151 private function parseHtml5(string $htmlContent, string $charset = 'UTF-8'): \DOMDocument
1152 {
1153 if (!$this->supportsEncoding($charset)) {
1154 $htmlContent = $this->convertToHtmlEntities($htmlContent, $charset);
1155 $charset = 'UTF-8';
1156 }
1157
1158 return $this->html5Parser->parse($htmlContent, ['encoding' => $charset]);
1159 }
1160
1161 private function supportsEncoding(string $encoding): bool
1162 {
1163 try {
1164 return '' === @mb_convert_encoding('', $encoding, 'UTF-8');
1165 } catch (\Throwable $e) {
1166 return false;
1167 }
1168 }
1169
1170 private function parseXhtml(string $htmlContent, string $charset = 'UTF-8'): \DOMDocument
1171 {
1172 if ('UTF-8' === $charset && preg_match('//u', $htmlContent)) {
1173 $htmlContent = '<?xml encoding="UTF-8">'.$htmlContent;
1174 } else {
1175 $htmlContent = $this->convertToHtmlEntities($htmlContent, $charset);
1176 }
1177
1178 $internalErrors = libxml_use_internal_errors(true);
1179 if (\LIBXML_VERSION < 20900) {
1180 $disableEntities = libxml_disable_entity_loader(true);
1181 }
1182
1183 $dom = new \DOMDocument('1.0', $charset);
1184 $dom->validateOnParse = true;
1185
1186 if ('' !== trim($htmlContent)) {
1187 @$dom->loadHTML($htmlContent);
1188 }
1189
1190 libxml_use_internal_errors($internalErrors);
1191 if (\LIBXML_VERSION < 20900) {
1192 libxml_disable_entity_loader($disableEntities);
1193 }
1194
1195 return $dom;
1196 }
1197
1198 /**
1199 * Converts charset to HTML-entities to ensure valid parsing.
1200 */
1201 private function convertToHtmlEntities(string $htmlContent, string $charset = 'UTF-8'): string
1202 {
1203 set_error_handler(function () { throw new \Exception(); });
1204
1205 try {
1206 return mb_encode_numericentity($htmlContent, [0x80, 0x10FFFF, 0, 0x1FFFFF], $charset);
1207 } catch (\Exception|\ValueError $e) {
1208 try {
1209 $htmlContent = iconv($charset, 'UTF-8', $htmlContent);
1210 $htmlContent = mb_encode_numericentity($htmlContent, [0x80, 0x10FFFF, 0, 0x1FFFFF], 'UTF-8');
1211 } catch (\Exception|\ValueError $e) {
1212 }
1213
1214 return $htmlContent;
1215 } finally {
1216 restore_error_handler();
1217 }
1218 }
1219
1220 /**
1221 * @throws \InvalidArgumentException
1222 */
1223 private function createDOMXPath(\DOMDocument $document, array $prefixes = []): \DOMXPath
1224 {
1225 $domxpath = new \DOMXPath($document);
1226
1227 foreach ($prefixes as $prefix) {
1228 $namespace = $this->discoverNamespace($domxpath, $prefix);
1229 if (null !== $namespace) {
1230 $domxpath->registerNamespace($prefix, $namespace);
1231 }
1232 }
1233
1234 return $domxpath;
1235 }
1236
1237 /**
1238 * @throws \InvalidArgumentException
1239 */
1240 private function discoverNamespace(\DOMXPath $domxpath, string $prefix): ?string
1241 {
1242 if (\array_key_exists($prefix, $this->namespaces)) {
1243 return $this->namespaces[$prefix];
1244 }
1245
1246 if ($this->cachedNamespaces->offsetExists($prefix)) {
1247 return $this->cachedNamespaces[$prefix];
1248 }
1249
1250 // ask for one namespace, otherwise we'd get a collection with an item for each node
1251 $namespaces = $domxpath->query(sprintf('(//namespace::*[name()="%s"])[last()]', $this->defaultNamespacePrefix === $prefix ? '' : $prefix));
1252
1253 return $this->cachedNamespaces[$prefix] = ($node = $namespaces->item(0)) ? $node->nodeValue : null;
1254 }
1255
1256 private function findNamespacePrefixes(string $xpath): array
1257 {
1258 if (preg_match_all('/(?P<prefix>[a-z_][a-z_0-9\-\.]*+):[^"\/:]/i', $xpath, $matches)) {
1259 return array_unique($matches['prefix']);
1260 }
1261
1262 return [];
1263 }
1264
1265 /**
1266 * Creates a crawler for some subnodes.
1267 *
1268 * @param \DOMNodeList|\DOMNode|\DOMNode[]|string|null $nodes
1269 *
1270 * @return static
1271 */
1272 private function createSubCrawler($nodes): object
1273 {
1274 $crawler = new static($nodes, $this->uri, $this->baseHref);
1275 $crawler->isHtml = $this->isHtml;
1276 $crawler->document = $this->document;
1277 $crawler->namespaces = $this->namespaces;
1278 $crawler->cachedNamespaces = $this->cachedNamespaces;
1279 $crawler->html5Parser = $this->html5Parser;
1280
1281 return $crawler;
1282 }
1283
1284 /**
1285 * @throws \LogicException If the CssSelector Component is not available
1286 */
1287 private function createCssSelectorConverter(): CssSelectorConverter
1288 {
1289 if (!class_exists(CssSelectorConverter::class)) {
1290 throw new \LogicException('To filter with a CSS selector, install the CssSelector component ("composer require symfony/css-selector"). Or use filterXpath instead.');
1291 }
1292
1293 return new CssSelectorConverter($this->isHtml);
1294 }
1295
1296 /**
1297 * Parse string into DOMDocument object using HTML5 parser if the content is HTML5 and the library is available.
1298 * Use libxml parser otherwise.
1299 */
1300 private function parseHtmlString(string $content, string $charset): \DOMDocument
1301 {
1302 if ($this->canParseHtml5String($content)) {
1303 return $this->parseHtml5($content, $charset);
1304 }
1305
1306 return $this->parseXhtml($content, $charset);
1307 }
1308
1309 private function canParseHtml5String(string $content): bool
1310 {
1311 if (null === $this->html5Parser) {
1312 return false;
1313 }
1314 if (false === ($pos = stripos($content, '<!doctype html>'))) {
1315 return false;
1316 }
1317 $header = substr($content, 0, $pos);
1318
1319 return '' === $header || $this->isValidHtml5Heading($header);
1320 }
1321
1322 private function isValidHtml5Heading(string $heading): bool
1323 {
1324 return 1 === preg_match('/^\x{FEFF}?\s*(<!--[^>]*?-->\s*)*$/u', $heading);
1325 }
1326 }
1327