PluginProbe
MxChat – AI Chatbot & Content Generation for WordPress / 2.2.6
MxChat – AI Chatbot & Content Generation for WordPress v2.2.6
3.2.21 3.2.20 3.2.19 3.2.18 3.2.17 3.2.16 3.2.15 3.2.14 3.2.12 3.2.13 3.2.11 3.2.10 3.2.9 3.2.8 3.2.7 3.2.6 3.2.5 3.2.4 3.2.3 3.2.2 3.2.1 2.0.3 2.0.4 2.0.5 2.0.6 All 152 releases
mxchat-basic / includes / pdf-parser / src / Smalot / PdfParser / Page.php

Page.php in MxChat – AI Chatbot & Content Generation for WordPress 2.2.6, at includes/pdf-parser/src/Smalot/PdfParser/Page.php

1,015 lines 37.1 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2
3 /**
4 * @file
5 * This file is part of the PdfParser library.
6 *
7 * @author Sébastien MALOT <sebastien@malot.fr>
8 *
9 * @date 2017-01-03
10 *
11 * @license LGPLv3
12 *
13 * @url <https://github.com/smalot/pdfparser>
14 *
15 * PdfParser is a pdf library written in PHP, extraction oriented.
16 * Copyright (C) 2017 - Sébastien MALOT <sebastien@malot.fr>
17 *
18 * This program is free software: you can redistribute it and/or modify
19 * it under the terms of the GNU Lesser General Public License as published by
20 * the Free Software Foundation, either version 3 of the License, or
21 * (at your option) any later version.
22 *
23 * This program is distributed in the hope that it will be useful,
24 * but WITHOUT ANY WARRANTY; without even the implied warranty of
25 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
26 * GNU Lesser General Public License for more details.
27 *
28 * You should have received a copy of the GNU Lesser General Public License
29 * along with this program.
30 * If not, see <http://www.pdfparser.org/sites/default/LICENSE.txt>.
31 */
32
33 namespace Smalot\PdfParser;
34
35 use Smalot\PdfParser\Element\ElementArray;
36 use Smalot\PdfParser\Element\ElementMissing;
37 use Smalot\PdfParser\Element\ElementNull;
38 use Smalot\PdfParser\Element\ElementXRef;
39
40 class Page extends PDFObject
41 {
42 /**
43 * @var Font[]
44 */
45 protected $fonts;
46
47 /**
48 * @var PDFObject[]
49 */
50 protected $xobjects;
51
52 /**
53 * @var array
54 */
55 protected $dataTm;
56
57 /**
58 * @param array<\Smalot\PdfParser\Font> $fonts
59 *
60 * @internal
61 */
62 public function setFonts($fonts)
63 {
64 if (empty($this->fonts)) {
65 $this->fonts = $fonts;
66 }
67 }
68
69 /**
70 * @return Font[]
71 */
72 public function getFonts()
73 {
74 if (null !== $this->fonts) {
75 return $this->fonts;
76 }
77
78 $resources = $this->get('Resources');
79
80 if (method_exists($resources, 'has') && $resources->has('Font')) {
81 if ($resources->get('Font') instanceof ElementMissing) {
82 return [];
83 }
84
85 if ($resources->get('Font') instanceof Header) {
86 $fonts = $resources->get('Font')->getElements();
87 } else {
88 $fonts = $resources->get('Font')->getHeader()->getElements();
89 }
90
91 $table = [];
92
93 foreach ($fonts as $id => $font) {
94 if ($font instanceof Font) {
95 $table[$id] = $font;
96
97 // Store too on cleaned id value (only numeric)
98 $id = preg_replace('/[^0-9\.\-_]/', '', $id);
99 if ('' != $id) {
100 $table[$id] = $font;
101 }
102 }
103 }
104
105 return $this->fonts = $table;
106 }
107
108 return [];
109 }
110
111 public function getFont(string $id): ?Font
112 {
113 $fonts = $this->getFonts();
114
115 if (isset($fonts[$id])) {
116 return $fonts[$id];
117 }
118
119 // According to the PDF specs (https://www.adobe.com/content/dam/acom/en/devnet/pdf/pdfs/PDF32000_2008.pdf, page 238)
120 // "The font resource name presented to the Tf operator is arbitrary, as are the names for all kinds of resources"
121 // Instead, we search for the unfiltered name first and then do this cleaning as a fallback, so all tests still pass.
122
123 if (isset($fonts[$id])) {
124 return $fonts[$id];
125 } else {
126 $id = preg_replace('/[^0-9\.\-_]/', '', $id);
127 if (isset($fonts[$id])) {
128 return $fonts[$id];
129 }
130 }
131
132 return null;
133 }
134
135 /**
136 * Support for XObject
137 *
138 * @return PDFObject[]
139 */
140 public function getXObjects()
141 {
142 if (null !== $this->xobjects) {
143 return $this->xobjects;
144 }
145
146 $resources = $this->get('Resources');
147
148 if (method_exists($resources, 'has') && $resources->has('XObject')) {
149 if ($resources->get('XObject') instanceof Header) {
150 $xobjects = $resources->get('XObject')->getElements();
151 } else {
152 $xobjects = $resources->get('XObject')->getHeader()->getElements();
153 }
154
155 $table = [];
156
157 foreach ($xobjects as $id => $xobject) {
158 $table[$id] = $xobject;
159
160 // Store too on cleaned id value (only numeric)
161 $id = preg_replace('/[^0-9\.\-_]/', '', $id);
162 if ('' != $id) {
163 $table[$id] = $xobject;
164 }
165 }
166
167 return $this->xobjects = $table;
168 }
169
170 return [];
171 }
172
173 public function getXObject(string $id): ?PDFObject
174 {
175 $xobjects = $this->getXObjects();
176
177 if (isset($xobjects[$id])) {
178 return $xobjects[$id];
179 }
180
181 return null;
182 /*$id = preg_replace('/[^0-9\.\-_]/', '', $id);
183
184 if (isset($xobjects[$id])) {
185 return $xobjects[$id];
186 } else {
187 return null;
188 }*/
189 }
190
191 public function getText(?self $page = null): string
192 {
193 if ($contents = $this->get('Contents')) {
194 if ($contents instanceof ElementMissing) {
195 return '';
196 } elseif ($contents instanceof ElementNull) {
197 return '';
198 } elseif ($contents instanceof PDFObject) {
199 $elements = $contents->getHeader()->getElements();
200
201 if (is_numeric(key($elements))) {
202 $new_content = '';
203
204 foreach ($elements as $element) {
205 if ($element instanceof ElementXRef) {
206 $new_content .= $element->getObject()->getContent();
207 } else {
208 $new_content .= $element->getContent();
209 }
210 }
211
212 $header = new Header([], $this->document);
213 $contents = new PDFObject($this->document, $header, $new_content, $this->config);
214 }
215 } elseif ($contents instanceof ElementArray) {
216 // Create a virtual global content.
217 $new_content = '';
218
219 foreach ($contents->getContent() as $content) {
220 $new_content .= $content->getContent()."\n";
221 }
222
223 $header = new Header([], $this->document);
224 $contents = new PDFObject($this->document, $header, $new_content, $this->config);
225 }
226
227 /*
228 * Elements referencing each other on the same page can cause endless loops during text parsing.
229 * To combat this we keep a recursionStack containing already parsed elements on the page.
230 * The stack is only emptied here after getting text from a page.
231 */
232 $contentsText = $contents->getText($this);
233 PDFObject::$recursionStack = [];
234
235 return $contentsText;
236 }
237
238 return '';
239 }
240
241 /**
242 * Return true if the current page is a (setasign\Fpdi\Fpdi) FPDI/FPDF document
243 *
244 * The metadata 'Producer' should have the value of "FPDF" . FPDF_VERSION if the
245 * pdf file was generated by FPDF/Fpfi.
246 *
247 * @return bool true is the current page is a FPDI/FPDF document
248 */
249 public function isFpdf(): bool
250 {
251 if (\array_key_exists('Producer', $this->document->getDetails())
252 && \is_string($this->document->getDetails()['Producer'])
253 && 0 === strncmp($this->document->getDetails()['Producer'], 'FPDF', 4)) {
254 return true;
255 }
256
257 return false;
258 }
259
260 /**
261 * Return the page number of the PDF document of the page object
262 *
263 * @return int the page number
264 */
265 public function getPageNumber(): int
266 {
267 $pages = $this->document->getPages();
268 $numOfPages = \count($pages);
269 for ($pageNum = 0; $pageNum < $numOfPages; ++$pageNum) {
270 if ($pages[$pageNum] === $this) {
271 break;
272 }
273 }
274
275 return $pageNum;
276 }
277
278 /**
279 * Return the Object of the page if the document is a FPDF/FPDI document
280 *
281 * If the document was generated by FPDF/FPDI it returns the
282 * PDFObject of the given page
283 *
284 * @return PDFObject The PDFObject for the page
285 */
286 public function getPDFObjectForFpdf(): PDFObject
287 {
288 $pageNum = $this->getPageNumber();
289 $xObjects = $this->getXObjects();
290
291 return $xObjects[$pageNum];
292 }
293
294 /**
295 * Return a new PDFObject of the document created with FPDF/FPDI
296 *
297 * For a document generated by FPDF/FPDI, it generates a
298 * new PDFObject for that document
299 *
300 * @return PDFObject The PDFObject
301 */
302 public function createPDFObjectForFpdf(): PDFObject
303 {
304 $pdfObject = $this->getPDFObjectForFpdf();
305 $new_content = $pdfObject->getContent();
306 $header = $pdfObject->getHeader();
307 $config = $pdfObject->config;
308
309 return new PDFObject($pdfObject->document, $header, $new_content, $config);
310 }
311
312 /**
313 * Return page if document is a FPDF/FPDI document
314 *
315 * @return Page The page
316 */
317 public function createPageForFpdf(): self
318 {
319 $pdfObject = $this->getPDFObjectForFpdf();
320 $new_content = $pdfObject->getContent();
321 $header = $pdfObject->getHeader();
322 $config = $pdfObject->config;
323
324 return new self($pdfObject->document, $header, $new_content, $config);
325 }
326
327 public function getTextArray(?self $page = null): array
328 {
329 if ($this->isFpdf()) {
330 $pdfObject = $this->getPDFObjectForFpdf();
331 $newPdfObject = $this->createPDFObjectForFpdf();
332
333 return $newPdfObject->getTextArray($pdfObject);
334 } else {
335 if ($contents = $this->get('Contents')) {
336 if ($contents instanceof ElementMissing) {
337 return [];
338 } elseif ($contents instanceof ElementNull) {
339 return [];
340 } elseif ($contents instanceof PDFObject) {
341 $elements = $contents->getHeader()->getElements();
342
343 if (is_numeric(key($elements))) {
344 $new_content = '';
345
346 /** @var PDFObject $element */
347 foreach ($elements as $element) {
348 if ($element instanceof ElementXRef) {
349 $new_content .= $element->getObject()->getContent();
350 } else {
351 $new_content .= $element->getContent();
352 }
353 }
354
355 $header = new Header([], $this->document);
356 $contents = new PDFObject($this->document, $header, $new_content, $this->config);
357 } else {
358 try {
359 $contents->getTextArray($this);
360 } catch (\Throwable $e) {
361 return $contents->getTextArray();
362 }
363 }
364 } elseif ($contents instanceof ElementArray) {
365 // Create a virtual global content.
366 $new_content = '';
367
368 /** @var PDFObject $content */
369 foreach ($contents->getContent() as $content) {
370 $new_content .= $content->getContent()."\n";
371 }
372
373 $header = new Header([], $this->document);
374 $contents = new PDFObject($this->document, $header, $new_content, $this->config);
375 }
376
377 return $contents->getTextArray($this);
378 }
379
380 return [];
381 }
382 }
383
384 /**
385 * Gets all the text data with its internal representation of the page.
386 *
387 * Returns an array with the data and the internal representation
388 */
389 public function extractRawData(): array
390 {
391 /*
392 * Now you can get the complete content of the object with the text on it
393 */
394 $extractedData = [];
395 $content = $this->get('Contents');
396 $values = $content->getContent();
397 if (isset($values) && \is_array($values)) {
398 $text = '';
399 foreach ($values as $section) {
400 $text .= $section->getContent();
401 }
402 $sectionsText = $this->getSectionsText($text);
403 foreach ($sectionsText as $sectionText) {
404 $commandsText = $this->getCommandsText($sectionText);
405 foreach ($commandsText as $command) {
406 $extractedData[] = $command;
407 }
408 }
409 } else {
410 if ($this->isFpdf()) {
411 $content = $this->getPDFObjectForFpdf();
412 }
413 $sectionsText = $content->getSectionsText($content->getContent());
414 foreach ($sectionsText as $sectionText) {
415 $commandsText = $content->getCommandsText($sectionText);
416 foreach ($commandsText as $command) {
417 $extractedData[] = $command;
418 }
419 }
420 }
421
422 return $extractedData;
423 }
424
425 /**
426 * Gets all the decoded text data with it internal representation from a page.
427 *
428 * @param array $extractedRawData the extracted data return by extractRawData or
429 * null if extractRawData should be called
430 *
431 * @return array An array with the data and the internal representation
432 */
433 public function extractDecodedRawData(?array $extractedRawData = null): array
434 {
435 if (!isset($extractedRawData) || !$extractedRawData) {
436 $extractedRawData = $this->extractRawData();
437 }
438 $currentFont = null; /** @var Font $currentFont */
439 $clippedFont = null;
440 $fpdfPage = null;
441 if ($this->isFpdf()) {
442 $fpdfPage = $this->createPageForFpdf();
443 }
444 foreach ($extractedRawData as &$command) {
445 if ('Tj' == $command['o'] || 'TJ' == $command['o']) {
446 $data = $command['c'];
447 if (!\is_array($data)) {
448 $tmpText = '';
449 if (isset($currentFont)) {
450 $tmpText = $currentFont->decodeOctal($data);
451 // $tmpText = $currentFont->decodeHexadecimal($tmpText, false);
452 }
453 $tmpText = str_replace(
454 ['\\\\', '\(', '\)', '\n', '\r', '\t', '\ '],
455 ['\\', '(', ')', "\n", "\r", "\t", ' '],
456 $tmpText
457 );
458 $tmpText = mb_convert_encoding($tmpText, 'UTF-8', 'ISO-8859-1');
459 if (isset($currentFont)) {
460 $tmpText = $currentFont->decodeContent($tmpText);
461 }
462 $command['c'] = $tmpText;
463 continue;
464 }
465 $numText = \count($data);
466 for ($i = 0; $i < $numText; ++$i) {
467 if (0 != ($i % 2)) {
468 continue;
469 }
470 $tmpText = $data[$i]['c'];
471 $decodedText = isset($currentFont) ? $currentFont->decodeOctal($tmpText) : $tmpText;
472 $decodedText = str_replace(
473 ['\\\\', '\(', '\)', '\n', '\r', '\t', '\ '],
474 ['\\', '(', ')', "\n", "\r", "\t", ' '],
475 $decodedText
476 );
477
478 $decodedText = mb_convert_encoding($decodedText, 'UTF-8', 'ISO-8859-1');
479
480 if (isset($currentFont)) {
481 $decodedText = $currentFont->decodeContent($decodedText);
482 }
483 $command['c'][$i]['c'] = $decodedText;
484 continue;
485 }
486 } elseif ('Tf' == $command['o'] || 'TF' == $command['o']) {
487 $fontId = explode(' ', $command['c'])[0];
488 // If document is a FPDI/FPDF the $page has the correct font
489 $currentFont = isset($fpdfPage) ? $fpdfPage->getFont($fontId) : $this->getFont($fontId);
490 continue;
491 } elseif ('Q' == $command['o']) {
492 $currentFont = $clippedFont;
493 } elseif ('q' == $command['o']) {
494 $clippedFont = $currentFont;
495 }
496 }
497
498 return $extractedRawData;
499 }
500
501 /**
502 * Gets just the Text commands that are involved in text positions and
503 * Text Matrix (Tm)
504 *
505 * It extract just the PDF commands that are involved with text positions, and
506 * the Text Matrix (Tm). These are: BT, ET, TL, Td, TD, Tm, T*, Tj, ', ", and TJ
507 *
508 * @param array $extractedDecodedRawData The data extracted by extractDecodeRawData.
509 * If it is null, the method extractDecodeRawData is called.
510 *
511 * @return array An array with the text command of the page
512 */
513 public function getDataCommands(?array $extractedDecodedRawData = null): array
514 {
515 if (!isset($extractedDecodedRawData) || !$extractedDecodedRawData) {
516 $extractedDecodedRawData = $this->extractDecodedRawData();
517 }
518 $extractedData = [];
519 foreach ($extractedDecodedRawData as $command) {
520 switch ($command['o']) {
521 /*
522 * BT
523 * Begin a text object, inicializind the Tm and Tlm to identity matrix
524 */
525 case 'BT':
526 $extractedData[] = $command;
527 break;
528 /*
529 * cm
530 * Concatenation Matrix that will transform all following Tm
531 */
532 case 'cm':
533 $extractedData[] = $command;
534 break;
535 /*
536 * ET
537 * End a text object, discarding the text matrix
538 */
539 case 'ET':
540 $extractedData[] = $command;
541 break;
542
543 /*
544 * leading TL
545 * Set the text leading, Tl, to leading. Tl is used by the T*, ' and " operators.
546 * Initial value: 0
547 */
548 case 'TL':
549 $extractedData[] = $command;
550 break;
551
552 /*
553 * tx ty Td
554 * Move to the start of the next line, offset form the start of the
555 * current line by tx, ty.
556 */
557 case 'Td':
558 $extractedData[] = $command;
559 break;
560
561 /*
562 * tx ty TD
563 * Move to the start of the next line, offset form the start of the
564 * current line by tx, ty. As a side effect, this operator set the leading
565 * parameter in the text state. This operator has the same effect as the
566 * code:
567 * -ty TL
568 * tx ty Td
569 */
570 case 'TD':
571 $extractedData[] = $command;
572 break;
573
574 /*
575 * a b c d e f Tm
576 * Set the text matrix, Tm, and the text line matrix, Tlm. The operands are
577 * all numbers, and the initial value for Tm and Tlm is the identity matrix
578 * [1 0 0 1 0 0]
579 */
580 case 'Tm':
581 $extractedData[] = $command;
582 break;
583
584 /*
585 * T*
586 * Move to the start of the next line. This operator has the same effect
587 * as the code:
588 * 0 Tl Td
589 * Where Tl is the current leading parameter in the text state.
590 */
591 case 'T*':
592 $extractedData[] = $command;
593 break;
594
595 /*
596 * string Tj
597 * Show a Text String
598 */
599 case 'Tj':
600 $extractedData[] = $command;
601 break;
602
603 /*
604 * string '
605 * Move to the next line and show a text string. This operator has the
606 * same effect as the code:
607 * T*
608 * string Tj
609 */
610 case "'":
611 $extractedData[] = $command;
612 break;
613
614 /*
615 * aw ac string "
616 * Move to the next lkine and show a text string, using aw as the word
617 * spacing and ac as the character spacing. This operator has the same
618 * effect as the code:
619 * aw Tw
620 * ac Tc
621 * string '
622 * Tw set the word spacing, Tw, to wordSpace.
623 * Tc Set the character spacing, Tc, to charsSpace.
624 */
625 case '"':
626 $extractedData[] = $command;
627 break;
628
629 case 'Tf':
630 case 'TF':
631 $extractedData[] = $command;
632 break;
633
634 /*
635 * array TJ
636 * Show one or more text strings allow individual glyph positioning.
637 * Each lement of array con be a string or a number. If the element is
638 * a string, this operator shows the string. If it is a number, the
639 * operator adjust the text position by that amount; that is, it translates
640 * the text matrix, Tm. This amount is substracted form the current
641 * horizontal or vertical coordinate, depending on the writing mode.
642 * in the default coordinate system, a positive adjustment has the effect
643 * of moving the next glyph painted either to the left or down by the given
644 * amount.
645 */
646 case 'TJ':
647 $extractedData[] = $command;
648 break;
649 /*
650 * q
651 * Save current graphics state to stack
652 */
653 case 'q':
654 /*
655 * Q
656 * Load last saved graphics state from stack
657 */
658 case 'Q':
659 $extractedData[] = $command;
660 break;
661 default:
662 }
663 }
664
665 return $extractedData;
666 }
667
668 /**
669 * Gets the Text Matrix of the text in the page
670 *
671 * Return an array where every item is an array where the first item is the
672 * Text Matrix (Tm) and the second is a string with the text data. The Text matrix
673 * is an array of 6 numbers. The last 2 numbers are the coordinates X and Y of the
674 * text. The first 4 numbers has to be with Scalation, Rotation and Skew of the text.
675 *
676 * @param array $dataCommands the data extracted by getDataCommands
677 * if null getDataCommands is called
678 *
679 * @return array an array with the data of the page including the Tm information
680 * of any text in the page
681 */
682 public function getDataTm(?array $dataCommands = null): array
683 {
684 if (!isset($dataCommands) || !$dataCommands) {
685 $dataCommands = $this->getDataCommands();
686 }
687
688 /*
689 * At the beginning of a text object Tm is the identity matrix
690 */
691 $defaultTm = ['1', '0', '0', '1', '0', '0'];
692 $concatTm = ['1', '0', '0', '1', '0', '0'];
693 $graphicsStatesStack = [];
694 /*
695 * Set the text leading used by T*, ' and " operators
696 */
697 $defaultTl = 0;
698
699 /*
700 * Set default values for font data
701 */
702 $defaultFontId = -1;
703 $defaultFontSize = 1;
704
705 /*
706 * Indexes of horizontal/vertical scaling and X,Y-coordinates in the matrix (Tm)
707 */
708 $hSc = 0; // horizontal scaling
709 /**
710 * index of vertical scaling in the array that encodes the text matrix.
711 * for more information: https://github.com/smalot/pdfparser/pull/559#discussion_r1053415500
712 */
713 $vSc = 3;
714 $x = 4;
715 $y = 5;
716
717 /*
718 * x,y-coordinates of text space origin in user units
719 *
720 * These will be assigned the value of the currently printed string
721 */
722 $Tx = 0;
723 $Ty = 0;
724
725 $Tm = $defaultTm;
726 $Tl = $defaultTl;
727 $fontId = $defaultFontId;
728 $fontSize = $defaultFontSize; // reflects fontSize set by Tf or Tfs
729
730 $extractedTexts = $this->getTextArray();
731 $extractedData = [];
732 foreach ($dataCommands as $command) {
733 // If we've used up all the texts from getTextArray(), exit
734 // so we aren't accessing non-existent array indices
735 // Fixes 'undefined array key' errors in Issues #575, #576
736 if (\count($extractedTexts) <= \count($extractedData)) {
737 break;
738 }
739 $currentText = $extractedTexts[\count($extractedData)];
740 switch ($command['o']) {
741 /*
742 * BT
743 * Begin a text object, initializing the Tm and Tlm to identity matrix
744 */
745 case 'BT':
746 $Tm = $defaultTm;
747 $Tl = $defaultTl;
748 $Tx = 0;
749 $Ty = 0;
750 break;
751
752 case 'cm':
753 $newConcatTm = (array) explode(' ', $command['c']);
754 $TempMatrix = [];
755 // Multiply with previous concatTm
756 $TempMatrix[0] = (float) $concatTm[0] * (float) $newConcatTm[0] + (float) $concatTm[1] * (float) $newConcatTm[2];
757 $TempMatrix[1] = (float) $concatTm[0] * (float) $newConcatTm[1] + (float) $concatTm[1] * (float) $newConcatTm[3];
758 $TempMatrix[2] = (float) $concatTm[2] * (float) $newConcatTm[0] + (float) $concatTm[3] * (float) $newConcatTm[2];
759 $TempMatrix[3] = (float) $concatTm[2] * (float) $newConcatTm[1] + (float) $concatTm[3] * (float) $newConcatTm[3];
760 $TempMatrix[4] = (float) $concatTm[4] * (float) $newConcatTm[0] + (float) $concatTm[5] * (float) $newConcatTm[2] + (float) $newConcatTm[4];
761 $TempMatrix[5] = (float) $concatTm[4] * (float) $newConcatTm[1] + (float) $concatTm[5] * (float) $newConcatTm[3] + (float) $newConcatTm[5];
762 $concatTm = $TempMatrix;
763 break;
764 /*
765 * ET
766 * End a text object
767 */
768 case 'ET':
769 break;
770
771 /*
772 * text leading TL
773 * Set the text leading, Tl, to leading. Tl is used by the T*, ' and " operators.
774 * Initial value: 0
775 */
776 case 'TL':
777 // scaled text leading
778 $Tl = (float) $command['c'] * (float) $Tm[$vSc];
779 break;
780
781 /*
782 * tx ty Td
783 * Move to the start of the next line, offset from the start of the
784 * current line by tx, ty.
785 */
786 case 'Td':
787 $coord = explode(' ', $command['c']);
788 $Tx += (float) $coord[0] * (float) $Tm[$hSc];
789 $Ty += (float) $coord[1] * (float) $Tm[$vSc];
790 $Tm[$x] = (string) $Tx;
791 $Tm[$y] = (string) $Ty;
792 break;
793
794 /*
795 * tx ty TD
796 * Move to the start of the next line, offset form the start of the
797 * current line by tx, ty. As a side effect, this operator set the leading
798 * parameter in the text state. This operator has the same effect as the
799 * code:
800 * -ty TL
801 * tx ty Td
802 */
803 case 'TD':
804 $coord = explode(' ', $command['c']);
805 $Tl = -((float) $coord[1] * (float) $Tm[$vSc]);
806 $Tx += (float) $coord[0] * (float) $Tm[$hSc];
807 $Ty += (float) $coord[1] * (float) $Tm[$vSc];
808 $Tm[$x] = (string) $Tx;
809 $Tm[$y] = (string) $Ty;
810 break;
811
812 /*
813 * a b c d e f Tm
814 * Set the text matrix, Tm, and the text line matrix, Tlm. The operands are
815 * all numbers, and the initial value for Tm and Tlm is the identity matrix
816 * [1 0 0 1 0 0]
817 */
818 case 'Tm':
819 $Tm = explode(' ', $command['c']);
820 $TempMatrix = [];
821 $TempMatrix[0] = (float) $Tm[0] * (float) $concatTm[0] + (float) $Tm[1] * (float) $concatTm[2];
822 $TempMatrix[1] = (float) $Tm[0] * (float) $concatTm[1] + (float) $Tm[1] * (float) $concatTm[3];
823 $TempMatrix[2] = (float) $Tm[2] * (float) $concatTm[0] + (float) $Tm[3] * (float) $concatTm[2];
824 $TempMatrix[3] = (float) $Tm[2] * (float) $concatTm[1] + (float) $Tm[3] * (float) $concatTm[3];
825 $TempMatrix[4] = (float) $Tm[4] * (float) $concatTm[0] + (float) $Tm[5] * (float) $concatTm[2] + (float) $concatTm[4];
826 $TempMatrix[5] = (float) $Tm[4] * (float) $concatTm[1] + (float) $Tm[5] * (float) $concatTm[3] + (float) $concatTm[5];
827 $Tm = $TempMatrix;
828 $Tx = (float) $Tm[$x];
829 $Ty = (float) $Tm[$y];
830 break;
831
832 /*
833 * T*
834 * Move to the start of the next line. This operator has the same effect
835 * as the code:
836 * 0 Tl Td
837 * Where Tl is the current leading parameter in the text state.
838 */
839 case 'T*':
840 $Ty -= $Tl;
841 $Tm[$y] = (string) $Ty;
842 break;
843
844 /*
845 * string Tj
846 * Show a Text String
847 */
848 case 'Tj':
849 $data = [$Tm, $currentText];
850 if ($this->config->getDataTmFontInfoHasToBeIncluded()) {
851 $data[] = $fontId;
852 $data[] = $fontSize;
853 }
854 $extractedData[] = $data;
855 break;
856
857 /*
858 * string '
859 * Move to the next line and show a text string. This operator has the
860 * same effect as the code:
861 * T*
862 * string Tj
863 */
864 case "'":
865 $Ty -= $Tl;
866 $Tm[$y] = (string) $Ty;
867 $extractedData[] = [$Tm, $currentText];
868 break;
869
870 /*
871 * aw ac string "
872 * Move to the next line and show a text string, using aw as the word
873 * spacing and ac as the character spacing. This operator has the same
874 * effect as the code:
875 * aw Tw
876 * ac Tc
877 * string '
878 * Tw set the word spacing, Tw, to wordSpace.
879 * Tc Set the character spacing, Tc, to charsSpace.
880 */
881 case '"':
882 $data = explode(' ', $currentText);
883 $Ty -= $Tl;
884 $Tm[$y] = (string) $Ty;
885 $extractedData[] = [$Tm, $data[2]]; // Verify
886 break;
887
888 case 'Tf':
889 /*
890 * From PDF 1.0 specification, page 106:
891 * fontname size Tf Set font and size
892 * Sets the text font and text size in the graphics state. There is no default value for
893 * either fontname or size; they must be selected using Tf before drawing any text.
894 * fontname is a resource name. size is a number expressed in text space units.
895 *
896 * Source: https://ia902503.us.archive.org/10/items/pdfy-0vt8s-egqFwDl7L2/PDF%20Reference%201.0.pdf
897 * Introduced with https://github.com/smalot/pdfparser/pull/516
898 */
899 list($fontId, $fontSize) = explode(' ', $command['c'], 2);
900 break;
901
902 /*
903 * array TJ
904 * Show one or more text strings allow individual glyph positioning.
905 * Each lement of array con be a string or a number. If the element is
906 * a string, this operator shows the string. If it is a number, the
907 * operator adjust the text position by that amount; that is, it translates
908 * the text matrix, Tm. This amount is substracted form the current
909 * horizontal or vertical coordinate, depending on the writing mode.
910 * in the default coordinate system, a positive adjustment has the effect
911 * of moving the next glyph painted either to the left or down by the given
912 * amount.
913 */
914 case 'TJ':
915 $data = [$Tm, $currentText];
916 if ($this->config->getDataTmFontInfoHasToBeIncluded()) {
917 $data[] = $fontId;
918 $data[] = $fontSize;
919 }
920 $extractedData[] = $data;
921 break;
922 /*
923 * q
924 * Save current graphics state to stack
925 */
926 case 'q':
927 $graphicsStatesStack[] = $concatTm;
928 break;
929 /*
930 * Q
931 * Load last saved graphics state from stack
932 */
933 case 'Q':
934 $concatTm = array_pop($graphicsStatesStack);
935 break;
936 default:
937 }
938 }
939 $this->dataTm = $extractedData;
940
941 return $extractedData;
942 }
943
944 /**
945 * Gets text data that are around the given coordinates (X,Y)
946 *
947 * If the text is in near the given coordinates (X,Y) (or the TM info),
948 * the text is returned. The extractedData return by getDataTm, could be use to see
949 * where is the coordinates of a given text, using the TM info for it.
950 *
951 * @param float $x The X value of the coordinate to search for. if null
952 * just the Y value is considered (same Row)
953 * @param float $y The Y value of the coordinate to search for
954 * just the X value is considered (same column)
955 * @param float $xError The value less or more to consider an X to be "near"
956 * @param float $yError The value less or more to consider an Y to be "near"
957 *
958 * @return array An array of text that are near the given coordinates. If no text
959 * "near" the x,y coordinate, an empty array is returned. If Both, x
960 * and y coordinates are null, null is returned.
961 */
962 public function getTextXY(?float $x = null, ?float $y = null, float $xError = 0, float $yError = 0): array
963 {
964 if (!isset($this->dataTm) || !$this->dataTm) {
965 $this->getDataTm();
966 }
967
968 if (null !== $x) {
969 $x = (float) $x;
970 }
971
972 if (null !== $y) {
973 $y = (float) $y;
974 }
975
976 if (null === $x && null === $y) {
977 return [];
978 }
979
980 $xError = (float) $xError;
981 $yError = (float) $yError;
982
983 $extractedData = [];
984 foreach ($this->dataTm as $item) {
985 $tm = $item[0];
986 $xTm = (float) $tm[4];
987 $yTm = (float) $tm[5];
988 $text = $item[1];
989 if (null === $y) {
990 if (($xTm >= ($x - $xError))
991 && ($xTm <= ($x + $xError))) {
992 $extractedData[] = [$tm, $text];
993 continue;
994 }
995 }
996 if (null === $x) {
997 if (($yTm >= ($y - $yError))
998 && ($yTm <= ($y + $yError))) {
999 $extractedData[] = [$tm, $text];
1000 continue;
1001 }
1002 }
1003 if (($xTm >= ($x - $xError))
1004 && ($xTm <= ($x + $xError))
1005 && ($yTm >= ($y - $yError))
1006 && ($yTm <= ($y + $yError))) {
1007 $extractedData[] = [$tm, $text];
1008 continue;
1009 }
1010 }
1011
1012 return $extractedData;
1013 }
1014 }
1015