PluginProbe
Templately – Elementor & Gutenberg Template Library: 6500+ Free & Pro Ready Templates And Cloud! / trunk
Templately – Elementor & Gutenberg Template Library: 6500+ Free & Pro Ready Templates And Cloud! vtrunk
3.8.0 3.7.5 3.7.4 3.7.3 3.7.2 1-final 3.7.1 3.7.0 3.6.8 3.6.7 3.6.6 3.6.5 3.6.4 3.6.3 3.6.2 3.6.1 3.0.3 3.0.4 3.0.5 3.0.6 3.0.7 3.0.8 3.0.9 3.1.0 3.1.1 All 112 releases
templately / modules / ai-content-merger / Utils / AIContentHelper.php

AIContentHelper.php in Templately – Elementor & Gutenberg Template Library: 6500+ Free & Pro Ready Templates And Cloud! trunk, at modules/ai-content-merger/Utils/AIContentHelper.php

1,198 lines 42.0 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2
3 namespace Templately\Modules\AiContentMerger\Utils;
4
5 /**
6 * AI Content Helper
7 *
8 * Standalone, self-contained AI content merger (spec 022 FR-011): platform-aware merging
9 * of AI-generated content into the original template — flatten-by-id, unicode
10 * normalization, Elementor/Gutenberg dot-notation merge, and HTML class-based replacement.
11 *
12 * Every method is a pure static with ZERO cross-module dependencies. Session/path
13 * resolution and file IO (reading the .ai.json / original JSON, writing the .ao.json debug
14 * file) are the CONSUMER's responsibility (FSI's Finalizer). The public entry point is the
15 * static merge() dispatcher.
16 */
17 class AIContentHelper {
18 /**
19 * Parity warnings accumulated during the most recent Gutenberg merge.
20 *
21 * A Gutenberg static block is validated by re-running its JS `save(attributes)` and
22 * comparing the output against the markup stored in the post. So a merge that writes new
23 * text into ONE of {attributes, markup} without the other produces a block the editor
24 * reports as "Block contains unexpected or invalid content" — and, because the stored
25 * markup is what the front end serves, a page that never shows the AI copy at all.
26 *
27 * Every replacement path below therefore moves both sides together. When one side cannot
28 * be updated (the old string is not found where it was expected — entity/encoding drift,
29 * an unknown block shape), that is recorded here instead of failing silently, so the
30 * Finalizer can log it and the post-import rebuild pass knows which posts to visit.
31 *
32 * @var array<int,array{blockName:string,blockId:string,attribute:string,reason:string}>
33 */
34 private static $parity_warnings = [];
35
36 public $htmlSources = [
37 'testimonial',
38 'feature-list',
39 'notice',
40 'pricing-table',
41 'typing-text',
42 'interactive-promo',
43 'call-to-action'
44 ];
45
46 /**
47 * Merge AI-generated content into the original template for a given platform.
48 *
49 * Pure, self-contained dispatcher (spec 022 FR-011): the caller supplies the already-read
50 * AI payload and original template; this routes to the platform-specific merger. Session
51 * resolution and file IO are the consumer's responsibility, not this class's.
52 *
53 * @param array $payload The decoded AI-generated template content.
54 * @param mixed $template The decoded original template to merge into.
55 * @param string $platform Either 'elementor' or 'gutenberg'.
56 * @return mixed The merged template, or the untouched $template for an unknown platform.
57 */
58 public static function merge( array $payload, $template, string $platform ) {
59 if ( 'elementor' === $platform ) {
60 return self::mergeAiContentWithOriginal( $payload, $template );
61 }
62 if ( 'gutenberg' === $platform ) {
63 return self::mergeAiContentWithOriginalGutenberg( $payload, $template );
64 }
65 return $template;
66 }
67
68 /**
69 * Normalize escape characters in AI content
70 *
71 * Removes backslashes before closing tags (converts `<\/` and `<\\/` to `</`)
72 * This normalization is applied to all content values in the AI template's contents arrays.
73 *
74 * @param array $flat Reference to the flattened AI content array
75 * @return void Modifies the array in place
76 */
77 public static function normalizeAiContentEscapeCharacters(&$flat) {
78 foreach ($flat as &$element) {
79 if (isset($element['contents']) && is_array($element['contents'])) {
80 foreach ($element['contents'] as $index => &$content_item) {
81 if (isset($content_item['content']) && is_string($content_item['content'])) {
82 // Normalize content: unescape closing tags (remove backslash before </)
83 $content_item['content'] = str_replace(['<\/', '<\\/'], '</', $content_item['content']);
84 }
85 else {
86 unset($element['contents'][$index]);
87 }
88 }
89 }
90 }
91 }
92
93 /**
94 * Recursively flatten a nested array by extracting elements with 'contents' and using their ID as key
95 *
96 * @param array $array The array to flatten
97 * @param array $flat Reference to the flattened array
98 * @return array The flattened array
99 */
100 public static function flattenById($array, &$flat = []) {
101 foreach ($array as $key => $value) {
102 if (is_array($value)) {
103 // If this is an element with 'widgetType' and 'contents', use its parent key as ID
104 if (isset($value['widgetType']) && isset($value['contents'])) {
105 $flat[$key] = $value;
106 }
107 // Recurse into children
108 self::flattenById($value, $flat);
109 }
110 }
111 return $flat;
112 }
113
114
115
116 /**
117 * Set a value in a nested array using a dot notation path
118 *
119 * @param array $array Reference to the array to modify
120 * @param array $path The path as an array of keys
121 * @param mixed $value The value to set
122 */
123 public static function setNestedValue(&$array, $path, $value) {
124 $key = array_shift($path);
125
126 if (empty($path)) {
127 // We've reached the final key, set the value
128 $array[$key] = $value;
129 } else {
130 // Initialize the nested array if it doesn't exist
131 if (!isset($array[$key]) || !is_array($array[$key])) {
132 $array[$key] = [];
133 }
134
135 // Continue recursively
136 self::setNestedValue($array[$key], $path, $value);
137 }
138 }
139
140 /**
141 * Merge AI content with the original template
142 *
143 * @param array $ai_template_json The AI template JSON
144 * @param array $original_template_json The original template JSON data
145 * @return array The merged template JSON
146 */
147 public static function mergeAiContentWithOriginal($ai_template_json, $original_template_json) {
148 // Some callers pass the AI template as a JSON-encoded string; decode it first.
149 if (is_string($ai_template_json)) {
150 $ai_template_json = json_decode($ai_template_json, true);
151 }
152
153 // 1. Flatten the AI template
154 $flat = self::flattenById($ai_template_json);
155
156 // 2. Normalize escape characters in AI content early in the pipeline
157 self::normalizeAiContentEscapeCharacters($flat);
158
159 $keys = array_keys($flat);
160
161 // 3. Loop through original content only once and update elements directly
162 self::updateElementorContentRecursively($flat, $keys, $original_template_json['content']);
163
164 return $original_template_json;
165 }
166
167 /**
168 * Update Elementor content recursively by looping through original content only once
169 *
170 * @param array $flat The flattened AI content array
171 * @param array $keys Array of element IDs from the flat array
172 * @param array $content Reference to the original content to update
173 */
174 public static function updateElementorContentRecursively($flat, array $keys, &$content) {
175 if (!is_array($content)) {
176 return;
177 }
178
179 // Check if this element has an ID and needs updating
180 if (isset($content['id']) && in_array($content['id'], $keys)) {
181 $element_id = $content['id'];
182 $element = $flat[$element_id];
183
184 if (isset($element['contents'])) {
185 // Update settings based on contents
186 foreach ($element['contents'] as $item) {
187 if (isset($item['attribute'], $item['content'])) {
188 $content_value = is_string($item['content'])
189 ? str_replace(['<\/', '<\\/'], '</', $item['content'])
190 : $item['content'];
191
192 // Support for dot notation in attribute paths
193 if (strpos($item['attribute'], '.') !== false) {
194 $path = explode('.', $item['attribute']);
195 self::setNestedValue($content['settings'], $path, $content_value);
196 } else {
197 $content['settings'][$item['attribute']] = $content_value;
198 }
199 }
200 }
201 }
202 }
203
204 // Recurse through elements array
205 if (isset($content['elements']) && is_array($content['elements'])) {
206 foreach ($content['elements'] as &$element) {
207 self::updateElementorContentRecursively($flat, $keys, $element);
208 }
209 }
210
211 // Recurse through all other array elements
212 foreach ($content as &$value) {
213 if (is_array($value)) {
214 self::updateElementorContentRecursively($flat, $keys, $value);
215 }
216 }
217 }
218
219 /**
220 * Merge AI content with the original Gutenberg template using advanced content replacement
221 *
222 * @param array $ai_template_json The AI template JSON
223 * @param array $original_template_json The original template JSON data
224 * @return array The merged template JSON
225 */
226 public static function mergeAiContentWithOriginalGutenberg($ai_template_json, $original_template_json) {
227 if (empty($original_template_json['content'])) {
228 return $original_template_json;
229 }
230
231 // Some callers pass the AI template as a JSON-encoded string; decode it first.
232 if (is_string($ai_template_json)) {
233 $ai_template_json = json_decode($ai_template_json, true);
234 }
235
236 self::$parity_warnings = [];
237
238 // 1. Flatten the AI template by block ID (for blocks with 'contents')
239 $flat = [];
240 self::flattenGutenbergById($ai_template_json, $flat);
241
242 // 2. Normalize escape characters in AI content early in the pipeline
243 self::normalizeAiContentEscapeCharacters($flat);
244
245 $generated = $flat;
246 $keys = array_keys($generated);
247
248 // 3. Parse the original Gutenberg content
249 $blocks = parse_blocks($original_template_json['content']);
250
251 // 4. Clean invalid blocks from parsed content
252 $blocks = self::cleanInvalidBlocks($blocks);
253
254 // 5. Replace content recursively using advanced replacer logic
255 $blocks = self::replaceGutenbergContentRecursively($generated, $keys, $blocks);
256
257 // 6. Clean invalid blocks before serialization
258 $blocks = self::cleanInvalidBlocks($blocks);
259
260 // 7. Serialize the updated blocks back to content
261 $original_template_json['content'] = serialize_blocks($blocks);
262
263 return $original_template_json;
264 }
265
266 /**
267 * Replace content recursively in Gutenberg blocks (ported from GutenbergContentReplacer)
268 */
269 public static function replaceGutenbergContentRecursively($generated, array $keys, &$blocks) {
270 $htmlSources = [
271 'testimonial',
272 'feature-list',
273 'notice',
274 'pricing-table',
275 'typing-text',
276 'interactive-promo',
277 'call-to-action'
278 ];
279
280 foreach ($blocks as &$block) {
281 if (!empty($block['attrs']['blockId'])) {
282 $blockId = $block['attrs']['blockId'];
283 if (in_array($blockId, $keys)) {
284 $blockData = $generated[$blockId];
285 $block_name = self::cleanBlockName( $block['blockName'] );
286
287 // Store old content BEFORE updating attributes
288 $oldContentMap = [];
289 if (!empty($blockData['contents']) && !in_array($block_name, $htmlSources)) {
290 foreach ($blockData['contents'] as $content) {
291 $attribute = $content['attribute'];
292 $oldContent = self::getNestedGutenbergAttribute($block['attrs'], $attribute);
293 if ($oldContent !== null) {
294 $oldContentMap[$attribute] = $oldContent;
295 }
296 }
297 }
298
299 // Replace content in attributes
300 if (!empty($blockData['contents']) && !in_array($block_name, $htmlSources)) {
301 foreach ($blockData['contents'] as $content) {
302 $attribute = $content['attribute'];
303 $newContent = $content['content'];
304 self::setNestedGutenbergAttribute($block['attrs'], $attribute, $newContent);
305 }
306 }
307
308 // Replace content in innerHTML and innerContent using old content
309 if (!empty($block['innerHTML']) || !empty($block['innerContent'])) {
310 self::replaceInGutenbergHtmlContent($block, $blockData, $oldContentMap);
311 }
312
313 if(in_array($block_name, $htmlSources)){
314 if (!empty($blockData['contents'])) {
315 // These blocks are addressed by CLASS NAME because that is where their copy
316 // actually lives: every block in $htmlSources declares its text attributes
317 // with a `source` + `selector` (EB's notice -> source:"text" on
318 // `.eb-notice-title`; feature-list -> source:"query" over the `li`s), so the
319 // values are PARSED BACK OUT OF THE MARKUP and never stored in the block
320 // comment. Rewriting the markup is therefore the complete fix for them.
321 //
322 // The attribute pass below is a safety net for the mixed case — a block that
323 // ALSO keeps a comment-stored copy of the same string, which would then go
324 // stale. Read the old text out of the markup FIRST: after the replacement it
325 // is gone, and it is the only key that can identify such an attribute (there
326 // is no class-name -> attribute map, and inventing one would rot on every EB
327 // markup change).
328 $oldByClass = self::extractContentByClassName(
329 is_string($block['innerHTML'] ?? null) ? $block['innerHTML'] : '',
330 $blockData['contents']
331 );
332
333 if (!empty($block['innerHTML'])) {
334 $block['innerHTML'] = self::replaceContentByClassName($block['innerHTML'], $blockData['contents']);
335 }
336 if (!empty($block['innerContent']) && is_array($block['innerContent'])) {
337 foreach ($block['innerContent'] as &$content) {
338 if (is_string($content)) {
339 $content = self::replaceContentByClassName($content, $blockData['contents']);
340 }
341 }
342 unset($content);
343 }
344
345 self::syncAttrsFromClassReplacements($block, $blockData['contents'], $oldByClass);
346 }
347 }
348 if($block["blockName"] === 'essential-blocks/accordion'){
349 $block_inner_block_ids = array_map(function($innerBlock) {
350 return $innerBlock["attrs"]["blockId"] ?? null;
351 }, $block["innerBlocks"]);
352 $_generated = array_fill_keys($block_inner_block_ids, ['contents' => $blockData['contents']]);
353
354 if(isset($block["innerBlocks"][0]["attrs"]["accordionLists"]) && count($block["innerBlocks"][0]["attrs"]["accordionLists"]) > 1){
355 $block["innerBlocks"] = self::replaceGutenbergContentRecursively($_generated, $block_inner_block_ids, $block['innerBlocks']);
356 }
357 else {
358 // The live path: EB syncs each accordion-item's `accordionLists` down from
359 // the parent as a SINGLE-entry array (accordion-item/src/edit.js), so the
360 // count > 1 branch above never fires for current EB markup.
361 //
362 // `accordion-item/src/save.js` renders `foundItem?.title` straight out of
363 // this attribute, so writing the updated entry here WITHOUT rewriting the
364 // item's own markup guarantees a save()/markup mismatch: the editor shows
365 // "Attempt recovery" on every FAQ row and the front end keeps serving the
366 // pre-AI titles. Move both sides together.
367 $attrAccordionLists = $block["attrs"]["accordionLists"];
368 $ids = array_column($attrAccordionLists, 'id');
369
370 foreach ($block["innerBlocks"] as $key => $accordion) {
371 $itemLists = $accordion["attrs"]["accordionLists"] ?? [];
372 if (!is_array($itemLists)) {
373 continue;
374 }
375
376 foreach($itemLists as $accordionKey => $accordionList){
377 if (!is_array($accordionList) || !isset($accordionList['id'])) {
378 continue;
379 }
380
381 // search $block["attrs"]["accordionLists"] by $accordionList["id"] and replace $accordionList with searched one
382 $foundIndex = array_search($accordionList["id"], $ids);
383 if ($foundIndex === false) {
384 continue;
385 }
386
387 $updatedEntry = $attrAccordionLists[$foundIndex];
388 $pairs = self::diffStringFields($accordionList, $updatedEntry);
389
390 $block["innerBlocks"][$key]["attrs"]["accordionLists"][$accordionKey] = $updatedEntry;
391
392 if (!empty($pairs)) {
393 self::applyReplacementsToBlockMarkup(
394 $block["innerBlocks"][$key],
395 self::buildReplacementRecords($pairs)
396 );
397 }
398 }
399 }
400 }
401 }
402 }
403
404 // Process nested blocks recursively
405 if (!empty($block['innerBlocks'])) {
406 $block['innerBlocks'] = self::replaceGutenbergContentRecursively($generated, $keys, $block['innerBlocks']);
407 }
408 }
409 }
410 return $blocks;
411 }
412
413 /**
414 * Parity warnings from the most recent Gutenberg merge.
415 *
416 * Empty means every replacement landed on both the attributes and the markup. A non-empty
417 * list names the blocks where one side could not be updated — those are exactly the blocks
418 * that will open with a recovery banner, so consumers (the FSI Finalizer log, the
419 * post-import rebuild queue) use it to decide what to report and what to revisit.
420 *
421 * @return array<int,array{blockName:string,blockId:string,attribute:string,reason:string}>
422 */
423 public static function collect_parity_warnings(): array {
424 return self::$parity_warnings;
425 }
426
427 /**
428 * Record one attribute/markup parity failure. Never throws — a merge that cannot keep the
429 * two sides in step must still produce content.
430 */
431 private static function addParityWarning($blockName, $blockId, $attribute, $reason) {
432 self::$parity_warnings[] = [
433 'blockName' => (string) $blockName,
434 'blockId' => (string) $blockId,
435 'attribute' => (string) $attribute,
436 'reason' => (string) $reason,
437 ];
438 }
439
440 /**
441 * String fields that differ between two versions of the same structured entry.
442 *
443 * Deliberately NOT limited to `title`: an accordion entry also carries
444 * `titlePrefixText`/`titleSuffixText`/`imageAlt`, and EB adds fields over time. Comparing
445 * every string field keeps this correct as the shape grows instead of silently covering
446 * one key.
447 *
448 * @param array $old Entry before the update.
449 * @param array $new Entry after the update.
450 * @return array<int,array{attribute:string,old:string,new:string}>
451 */
452 private static function diffStringFields($old, $new) {
453 $pairs = [];
454
455 if (!is_array($old) || !is_array($new)) {
456 return $pairs;
457 }
458
459 foreach ($new as $field => $value) {
460 if (!is_string($value) || !isset($old[$field]) || !is_string($old[$field])) {
461 continue;
462 }
463 if ($old[$field] === '' || $old[$field] === $value) {
464 continue;
465 }
466 $pairs[] = [
467 'attribute' => (string) $field,
468 'old' => $old[$field],
469 'new' => $value,
470 ];
471 }
472
473 return $pairs;
474 }
475
476 /**
477 * Shape old/new pairs into the replacement records `replaceGutenbergContentInHtml()` expects.
478 *
479 * Longest-first, for the same reason `replaceInGutenbergHtmlContent()` sorts: a short string
480 * that is a prefix of a longer one would otherwise consume it.
481 *
482 * @param array<int,array{attribute:string,old:string,new:string}> $pairs
483 * @return array
484 */
485 private static function buildReplacementRecords($pairs) {
486 $replacements = [];
487
488 foreach ($pairs as $pair) {
489 $replacements[] = [
490 'originalFormat' => $pair['old'],
491 'decodedFormat' => json_decode('"' . $pair['old'] . '"'),
492 'normalizedFormat' => self::normalizeGutenbergUnicodeContent($pair['old']),
493 'newContent' => $pair['new'],
494 'attribute' => $pair['attribute'],
495 ];
496 }
497
498 usort($replacements, function ($a, $b) {
499 return strlen($b['originalFormat']) - strlen($a['originalFormat']);
500 });
501
502 return $replacements;
503 }
504
505 /**
506 * Apply replacement records to a block's own markup (`innerHTML` + `innerContent`).
507 *
508 * Records a parity warning for any replacement whose old text is still present afterwards —
509 * that block's save() output will not match what is stored.
510 *
511 * @param array $block Block, by reference.
512 * @param array $replacements Records from {@see self::buildReplacementRecords()}.
513 */
514 private static function applyReplacementsToBlockMarkup(&$block, $replacements) {
515 if (empty($replacements)) {
516 return;
517 }
518
519 $before = self::ownMarkup($block);
520
521 if (!empty($block['innerHTML']) && is_string($block['innerHTML'])) {
522 $block['innerHTML'] = self::replaceGutenbergContentInHtml($block['innerHTML'], $replacements);
523 }
524
525 if (!empty($block['innerContent']) && is_array($block['innerContent'])) {
526 foreach ($block['innerContent'] as $index => $chunk) {
527 if (is_string($chunk)) {
528 $block['innerContent'][$index] = self::replaceGutenbergContentInHtml($chunk, $replacements);
529 }
530 }
531 }
532
533 $after = self::ownMarkup($block);
534 $hasChildren = !empty($block['innerBlocks']);
535
536 foreach ($replacements as $replacement) {
537 $old = $replacement['originalFormat'];
538 $new = $replacement['newContent'];
539
540 if ($old === '' || $new === '') {
541 continue;
542 }
543
544 $wasHere = strpos($before, $old) !== false;
545
546 if ($wasHere) {
547 // The copy IS in this block's markup. If it survived the replacement the block is
548 // now desynced from its own attributes — this is the "Attempt recovery" case.
549 if (strpos($after, $old) !== false) {
550 self::addParityWarning($block['blockName'] ?? '', $block['attrs']['blockId'] ?? '', $replacement['attribute'], 'markup_not_updated');
551 }
552 continue;
553 }
554
555 // The copy was not in this block's own markup. `parse_blocks` gives each block only
556 // its OWN chunks, so for a container (an accordion parent, a wrapper) the text
557 // legitimately lives in a child block that is walked separately — silence there.
558 // With no children there is nowhere else for it to be, so the replacement had no
559 // target and the two sides cannot agree.
560 if (!$hasChildren && strpos($after, $new) === false && trim($after) !== '') {
561 self::addParityWarning($block['blockName'] ?? '', $block['attrs']['blockId'] ?? '', $replacement['attribute'], 'markup_not_updated');
562 }
563 }
564 }
565
566 /**
567 * A block's OWN markup — `innerHTML` plus its string `innerContent` chunks, never its
568 * children's (`parse_blocks` keeps those in `innerBlocks`, walked separately).
569 *
570 * @param array $block
571 * @return string
572 */
573 private static function ownMarkup($block) {
574 $markup = is_string($block['innerHTML'] ?? null) ? $block['innerHTML'] : '';
575
576 if (!empty($block['innerContent']) && is_array($block['innerContent'])) {
577 foreach ($block['innerContent'] as $chunk) {
578 if (is_string($chunk)) {
579 $markup .= $chunk;
580 }
581 }
582 }
583
584 return $markup;
585 }
586
587 /**
588 * Current text of each class-addressed target, keyed by the class name used to address it.
589 *
590 * Must be called BEFORE the class-based replacement runs — afterwards the old text is gone,
591 * and it is the only thing that can identify the attribute holding the same copy.
592 *
593 * @param string $html Block markup.
594 * @param array $contents `[['attribute' => className, 'content' => newContent], …]`
595 * @return array<string,string> className => old text
596 */
597 public static function extractContentByClassName($html, $contents) {
598 $found = [];
599
600 if (!is_string($html) || $html === '' || empty($contents)) {
601 return $found;
602 }
603 if (!class_exists('DOMDocument') || !class_exists('DOMXPath')) {
604 return $found;
605 }
606
607 $dom = new \DOMDocument();
608 @$dom->loadHTML('<?xml encoding="utf-8" ?>' . self::escapeInvalidEntities($html), LIBXML_HTML_NOIMPLIED | LIBXML_HTML_NODEFDTD);
609 $xpath = new \DOMXPath($dom);
610
611 foreach ($contents as $item) {
612 if (!isset($item['attribute'])) {
613 continue;
614 }
615 $className = $item['attribute'];
616 $baseClassName = self::extractBaseClassName($className);
617 $targetIndex = self::extractClassIndex($className);
618
619 $nodes = $xpath->query("//*[contains(concat(' ', normalize-space(@class), ' '), ' $baseClassName ')]");
620 if (!$nodes || $nodes->length === 0) {
621 continue;
622 }
623
624 $node = $targetIndex !== null ? ($nodes[$targetIndex] ?? null) : $nodes[0];
625 if ($node !== null) {
626 $found[$className] = $node->nodeValue;
627 }
628 }
629
630 return $found;
631 }
632
633 /**
634 * Write class-replaced copy back into the block's attributes, matched BY VALUE.
635 *
636 * For the `$htmlSources` blocks the copy normally lives ONLY in the markup (their text
637 * attributes are `source`d from selectors), so this usually finds nothing — that is the
638 * expected, healthy case and is deliberately NOT warned about. It exists for the mixed
639 * block that also keeps a comment-stored copy of the same string, which would otherwise go
640 * stale and make save() disagree with the markup.
641 *
642 * Matching on the exact old string needs no class-name -> attribute table and cannot rot
643 * when EB renames a class; a table would have to be maintained per block per release, and a
644 * stale entry fails silently.
645 *
646 * Exact FULL-string equality only (never substring), and only on string scalars, so an
647 * unrelated attribute that merely contains the text is left alone.
648 *
649 * @param array $block Block, by reference.
650 * @param array $contents `[['attribute' => className, 'content' => newContent], …]`
651 * @param array $oldByClass className => old text, from {@see self::extractContentByClassName()}
652 */
653 private static function syncAttrsFromClassReplacements(&$block, $contents, $oldByClass) {
654 if (empty($contents) || empty($oldByClass) || empty($block['attrs']) || !is_array($block['attrs'])) {
655 return;
656 }
657
658 $map = [];
659 foreach ($contents as $item) {
660 if (!isset($item['attribute'], $item['content']) || !is_string($item['content'])) {
661 continue;
662 }
663 $className = $item['attribute'];
664 if (!isset($oldByClass[$className])) {
665 continue;
666 }
667
668 $old = $oldByClass[$className];
669 if (!is_string($old)) {
670 continue;
671 }
672
673 $old = trim($old);
674 if ($old === '' || $old === $item['content']) {
675 continue;
676 }
677
678 $map[$old] = ['new' => $item['content'], 'attribute' => $className];
679 }
680
681 if (empty($map)) {
682 return;
683 }
684
685 $written = [];
686 self::replaceStringsInAttrs($block['attrs'], $map, $written);
687 }
688
689 /**
690 * Recursively replace exact string values inside an attribute tree.
691 *
692 * @param mixed $attrs Attribute value/tree, by reference.
693 * @param array $map oldString => ['new' => newString, 'attribute' => label]
694 * @param array $written oldString => hit count, by reference.
695 */
696 private static function replaceStringsInAttrs(&$attrs, $map, &$written) {
697 if (is_string($attrs)) {
698 $key = trim($attrs);
699 if (isset($map[$key])) {
700 $attrs = $map[$key]['new'];
701 $written[$key] = ($written[$key] ?? 0) + 1;
702 }
703 return;
704 }
705
706 if (is_array($attrs)) {
707 foreach ($attrs as $key => $value) {
708 self::replaceStringsInAttrs($attrs[$key], $map, $written);
709 }
710 }
711 }
712
713 /**
714 * Set nested attribute value using dot notation (ported from GutenbergContentReplacer)
715 */
716 public static function setNestedGutenbergAttribute(&$attrs, $path, $value) {
717 $keys = explode('.', $path);
718 $current = &$attrs;
719 for ($i = 0; $i < count($keys) - 1; $i++) {
720 $key = $keys[$i];
721 if (!isset($current[$key])) {
722 $current[$key] = [];
723 }
724 $current = &$current[$key];
725 }
726 $finalKey = end($keys);
727 $current[$finalKey] = $value;
728 }
729
730 /**
731 * Get nested attribute value using dot notation (ported from GutenbergContentReplacer)
732 */
733 public static function getNestedGutenbergAttribute($attrs, $path) {
734 $keys = explode('.', $path);
735 $current = $attrs;
736 foreach ($keys as $key) {
737 if (!isset($current[$key])) {
738 return null;
739 }
740 $current = $current[$key];
741 }
742 return $current;
743 }
744
745 /**
746 * Replace content in innerHTML and innerContent while preserving HTML structure (ported from GutenbergContentReplacer)
747 */
748 public static function replaceInGutenbergHtmlContent(&$block, $blockData, $oldContentMap) {
749 if (empty($blockData['contents']) || empty($oldContentMap)) return;
750 $replacements = [];
751 foreach ($blockData['contents'] as $content) {
752 $attribute = $content['attribute'];
753 $newContent = $content['content'];
754 if (isset($oldContentMap[$attribute])) {
755 $oldAttributeContent = $oldContentMap[$attribute];
756 $decodedUnicodeContent = json_decode('"' . $oldAttributeContent . '"');
757 $normalizedAttributeContent = self::normalizeGutenbergUnicodeContent($oldAttributeContent);
758 $normalizedNewContent = self::normalizeGutenbergUnicodeContent($newContent);
759 if ($normalizedAttributeContent !== $normalizedNewContent) {
760 $replacements[] = [
761 'originalFormat' => $oldAttributeContent,
762 'decodedFormat' => $decodedUnicodeContent,
763 'normalizedFormat' => $normalizedAttributeContent,
764 'newContent' => $newContent,
765 'attribute' => $attribute
766 ];
767 }
768 }
769 }
770 // sort $replacements by length of 'originalFormat' in descending order
771 usort($replacements, function($a, $b) {
772 return strlen($b['originalFormat']) - strlen($a['originalFormat']);
773 });
774 if (empty($replacements)) {
775 return;
776 }
777
778 // One call, so the leftover check below sees the SAME markup the block ends up with —
779 // the attributes have already been rewritten by the caller, so anything this fails to
780 // replace is a desync that will surface as "Attempt recovery" in the editor.
781 self::applyReplacementsToBlockMarkup($block, $replacements);
782 }
783
784 /**
785 * Replace content in HTML while preserving structure and handling Unicode (ported from GutenbergContentReplacer)
786 *
787 * Uses targeted replacement that avoids replacing text inside HTML attributes (href, src, data-*, etc.)
788 * to prevent breaking URLs and other attribute values.
789 */
790 public static function replaceGutenbergContentInHtml($html, $replacements) {
791 foreach ($replacements as $replacement) {
792 $originalFormat = $replacement['originalFormat'];
793 $decodedFormat = $replacement['decodedFormat'];
794 $normalizedFormat = $replacement['normalizedFormat'];
795 $newContent = $replacement['newContent'];
796 if (empty($originalFormat)) continue;
797 $htmlNewContent = $newContent;
798
799 // Use targeted replacement that avoids HTML attributes
800 $html = self::replaceTextOutsideAttributes($html, $originalFormat, $htmlNewContent);
801
802 if ($decodedFormat !== null && $decodedFormat !== $originalFormat) {
803 $html = self::replaceTextOutsideAttributes($html, $decodedFormat, $htmlNewContent);
804 }
805 if ($normalizedFormat !== $decodedFormat && $normalizedFormat !== $originalFormat) {
806 $html = self::replaceTextOutsideAttributes($html, $normalizedFormat, $htmlNewContent);
807 }
808 }
809 return $html;
810 }
811
812 /**
813 * Replace text in HTML only outside of HTML tags and attributes
814 *
815 * This function replaces occurrences of $oldText with $newText, but only when the text
816 * appears outside of HTML tags and attributes. This prevents unintended replacements
817 * inside URLs, src attributes, href attributes, and other HTML attributes.
818 *
819 * @param string $html The HTML content to process
820 * @param string $oldText The text to find and replace
821 * @param string $newText The replacement text
822 * @return string The HTML with replacements applied only outside of tags/attributes
823 */
824 private static function replaceTextOutsideAttributes($html, $oldText, $newText) {
825 if (empty($oldText) || $oldText === $newText) {
826 return $html;
827 }
828
829 $result = '';
830 $lastPos = 0;
831
832 // Find all occurrences of the text
833 while (($pos = strpos($html, $oldText, $lastPos)) !== false) {
834 // Check if this occurrence is inside an HTML tag or attribute
835 if (!self::isPositionInsideTag($html, $pos)) {
836 // Not inside a tag, safe to replace
837 $result .= substr($html, $lastPos, $pos - $lastPos) . $newText;
838 $lastPos = $pos + strlen($oldText);
839 } else {
840 // Inside a tag, skip this occurrence
841 $result .= substr($html, $lastPos, $pos - $lastPos + strlen($oldText));
842 $lastPos = $pos + strlen($oldText);
843 }
844 }
845
846 // Append remaining HTML
847 $result .= substr($html, $lastPos);
848 return $result;
849 }
850
851 /**
852 * Check if a position in HTML is inside an HTML attribute value
853 *
854 * This checks if the position is between quotes within an HTML tag.
855 * Returns true only if the position is inside an attribute value (between quotes),
856 * not just anywhere inside a tag.
857 *
858 * @param string $html The HTML content
859 * @param int $position The position to check
860 * @return bool True if the position is inside an attribute value, false otherwise
861 */
862 private static function isPositionInsideTag($html, $position) {
863 // Get the text before the position
864 $beforeText = substr($html, 0, $position);
865
866 // Find the last < and > before the position
867 $lastOpenTag = strrpos($beforeText, '<');
868 $lastCloseTag = strrpos($beforeText, '>');
869
870 // If there's no unclosed tag, we're not inside a tag
871 if ($lastOpenTag === false || ($lastCloseTag !== false && $lastOpenTag < $lastCloseTag)) {
872 return false;
873 }
874
875 // We're inside a tag. Now check if we're inside an attribute value (between quotes)
876 // Get the tag content from the last < to the position
877 $tagContent = substr($html, $lastOpenTag, $position - $lastOpenTag);
878
879 // Track whether we're inside double or single quotes by iterating through the tag content
880 $inDoubleQuotes = false;
881 $inSingleQuotes = false;
882
883 for ($i = 0; $i < strlen($tagContent); $i++) {
884 $char = $tagContent[$i];
885
886 // Toggle quote state when we encounter a quote
887 if ($char === '"' && !$inSingleQuotes) {
888 $inDoubleQuotes = !$inDoubleQuotes;
889 } elseif ($char === "'" && !$inDoubleQuotes) {
890 $inSingleQuotes = !$inSingleQuotes;
891 }
892 }
893
894 // We're inside an attribute value if we're inside either type of quotes
895 return $inDoubleQuotes || $inSingleQuotes;
896 }
897
898 /**
899 * Normalize Unicode content to handle different apostrophe types and other Unicode variations (ported from GutenbergContentReplacer)
900 */
901 public static function normalizeGutenbergUnicodeContent($content) {
902 $decoded = json_decode('"' . $content . '"');
903 if ($decoded !== null) {
904 $content = $decoded;
905 }
906 $unicodeReplacements = [
907 '\u2019' => "'",
908 '\u2018' => "'",
909 '\u201C' => '"',
910 '\u201D' => '"',
911 '\u2013' => '-',
912 '\u2014' => '-',
913 '\u2026' => '...',
914 "\u{2019}" => "'",
915 "\u{2018}" => "'",
916 "\u{201C}" => '"',
917 "\u{201D}" => '"',
918 "\u{2013}" => '-',
919 "\u{2014}" => '-',
920 "\u{2026}" => '...'
921 ];
922 return str_replace(array_keys($unicodeReplacements), array_values($unicodeReplacements), $content);
923 }
924
925 /**
926 * Convert content to HTML format (handle line breaks and inline tags) (ported from GutenbergContentReplacer)
927 */
928 public static function convertGutenbergToHtmlFormat($content) {
929 $content = str_replace("\n", '<br>', $content);
930 $content = str_replace("\r\n", '<br>', $content);
931 return $content;
932 }
933
934 /**
935 * Recursively flatten a nested Gutenberg AI array by extracting blocks with 'contents' and using their blockId as key
936 *
937 * @param array $array The array to flatten
938 * @param array $flat Reference to the flattened array
939 * @return array The flattened array
940 */
941 public static function flattenGutenbergById($array, &$flat = []) {
942 foreach ($array as $key => $value) {
943 if (is_array($value)) {
944 // If this is a block with 'blockName' and 'contents', use its parent key as ID
945 if (isset($value['blockName']) && isset($value['contents'])) {
946 $flat[$key] = $value;
947 }
948 // Recurse into children
949 self::flattenGutenbergById($value, $flat);
950 }
951 }
952 return $flat;
953 }
954
955 /**
956 * Replace the inner content of tags with given class names in the HTML.
957 * Supports indexed class names (e.g., "eb-feature-list-title.0", "eb-feature-list-title.1").
958 * Falls back to regex if DOMDocument does not find the class.
959 *
960 * @param string $html The HTML string.
961 * @param array $contents Array of ['attribute' => className, 'content' => newContent]
962 * @return string The updated HTML.
963 */
964 public static function replaceContentByClassName($html, $contents) {
965 $classExists = false;
966 foreach ($contents as $item) {
967 $className = $item['attribute'];
968 // Extract base class name (remove index if present)
969 $baseClassName = self::extractBaseClassName($className);
970 if (preg_match('/class=["\'][^"\']*\b' . preg_quote($baseClassName, '/') . '\b[^"\']*["\']/', $html)) {
971 $classExists = true;
972 break;
973 }
974 }
975 if (!$classExists) {
976 return $html; // No relevant class found, skip both methods
977 }
978
979 if (class_exists('DOMDocument') && class_exists('DOMXPath')) {
980 return self::replaceContentByClassNameDom($html, $contents);
981 } else {
982 return self::replaceContentByClassNameRegex($html, $contents);
983 }
984 }
985
986 /**
987 * Extract base class name from indexed class name.
988 *
989 * @param string $className The class name (e.g., "eb-feature-list-title.0")
990 * @return string The base class name (e.g., "eb-feature-list-title")
991 */
992 public static function extractBaseClassName($className) {
993 // Check if class name has numeric index at the end
994 if (preg_match('/^(.+)\.(\d+)$/', $className, $matches)) {
995 return $matches[1]; // Return base class name
996 }
997 return $className; // Return original if no index found
998 }
999
1000 /**
1001 * Extract index from indexed class name.
1002 *
1003 * @param string $className The class name (e.g., "eb-feature-list-title.0")
1004 * @return int|null The index (e.g., 0) or null if no index found
1005 */
1006 public static function extractClassIndex($className) {
1007 // Check if class name has numeric index at the end
1008 if (preg_match('/^(.+)\.(\d+)$/', $className, $matches)) {
1009 return (int)$matches[2]; // Return index as integer
1010 }
1011 return null; // Return null if no index found
1012 }
1013
1014 /**
1015 * Replace the inner content of tags with given class names in the HTML using DOMDocument.
1016 * Supports indexed class names (e.g., "eb-feature-list-title.0", "eb-feature-list-title.1").
1017 *
1018 * Note: While CSS selectors would be more readable, PHP's DOMDocument doesn't natively support
1019 * CSS selectors. We use XPath which is the standard way to query DOM elements in PHP.
1020 * For CSS selector support, you would need a third-party library like symfony/css-selector
1021 * or QueryPath, but we keep this implementation dependency-free.
1022 *
1023 * @param string $html The HTML string.
1024 * @param array $contents Array of ['attribute' => className, 'content' => newContent]
1025 * @return string The updated HTML.
1026 */
1027 public static function replaceContentByClassNameDom($html, $contents) {
1028 $dom = new \DOMDocument();
1029 // Suppress errors due to HTML5 tags or fragments
1030 $html = self::escapeInvalidEntities($html);
1031 @$dom->loadHTML('<?xml encoding="utf-8" ?>' . $html, LIBXML_HTML_NOIMPLIED | LIBXML_HTML_NODEFDTD);
1032
1033 $xpath = new \DOMXPath($dom);
1034 foreach ($contents as $item) {
1035 $className = $item['attribute'];
1036 $newContent = self::escapeInvalidEntities($item['content']);
1037 // $newContent = $item['content'];
1038
1039 // Extract base class name and index
1040 $baseClassName = self::extractBaseClassName($className);
1041 $targetIndex = self::extractClassIndex($className);
1042
1043 // Find elements by base class name using XPath
1044 // XPath equivalent to CSS selector: .baseClassName
1045 $nodes = $xpath->query("//*[contains(concat(' ', normalize-space(@class), ' '), ' $baseClassName ')]");
1046
1047 if ($targetIndex !== null) {
1048 // If indexed, only replace the element at the specific index
1049 if (isset($nodes[$targetIndex])) {
1050 $nodes[$targetIndex]->nodeValue = $newContent;
1051 }
1052 } else {
1053 // If not indexed, replace all elements with the class
1054 foreach ($nodes as $node) {
1055 $node->nodeValue = $newContent;
1056 }
1057 }
1058 }
1059 // Remove the XML encoding declaration
1060 $result = $dom->saveHTML();
1061 $result = preg_replace('/^<\?xml.*?\?>/', '', $result);
1062 return $result;
1063 }
1064
1065 /**
1066 * Replace the inner content of tags with given class names in the HTML using regex.
1067 * Supports indexed class names (e.g., "eb-feature-list-title.0", "eb-feature-list-title.1").
1068 *
1069 * @param string $html The HTML string.
1070 * @param array $contents Array of ['attribute' => className, 'content' => newContent]
1071 * @return string The updated HTML.
1072 */
1073 public static function replaceContentByClassNameRegex($html, $contents) {
1074 foreach ($contents as $item) {
1075 $className = $item['attribute'];
1076 $newContent = $item['content'];
1077
1078 // Extract base class name and index
1079 $baseClassName = self::extractBaseClassName($className);
1080 $targetIndex = self::extractClassIndex($className);
1081
1082 if ($targetIndex !== null) {
1083 // Handle indexed replacement
1084 $html = self::replaceContentByClassNameRegexIndexed($html, $baseClassName, $newContent, $targetIndex);
1085 } else {
1086 // Handle non-indexed replacement (original behavior)
1087 $quotedClassName = preg_quote($className, '/');
1088 $pattern = '/(<([a-z0-9]+)[^>]*class="[^"]*\b' . $quotedClassName . '\b[^"]*"[^>]*>)(.*?)(<\/\2>)/is';
1089 $replacement = '$1' . $newContent . '$4';
1090 $html = preg_replace($pattern, $replacement, $html);
1091 }
1092 }
1093 return $html;
1094 }
1095
1096 /**
1097 * Replace content for a specific indexed occurrence of a class name using regex.
1098 *
1099 * @param string $html The HTML string.
1100 * @param string $baseClassName The base class name (without index).
1101 * @param string $newContent The new content to replace.
1102 * @param int $targetIndex The zero-based index of the element to replace.
1103 * @return string The updated HTML.
1104 */
1105 public static function replaceContentByClassNameRegexIndexed($html, $baseClassName, $newContent, $targetIndex) {
1106 $quotedClassName = preg_quote($baseClassName, '/');
1107 /*
1108 Regex explanation:
1109 - (<([a-z0-9]+)[^>]*class="[^"]*\b$baseClassName\b[^"]*"[^>]*>)
1110 - (<([a-z0-9]+)[^>]* ... >) : Captures the opening tag with any attributes
1111 - ([a-z0-9]+) : Captures the tag name (e.g., p, div, span)
1112 - class="[^"]*\b$baseClassName\b[^"]*" : Ensures the class attribute contains the exact base class name (word boundary)
1113 - (.*?) : Captures everything inside the tag (non-greedy)
1114 - (<\/\2>) : Matches the corresponding closing tag (\2 is the tag name from earlier)
1115 Flags:
1116 - i : case-insensitive (for tag names)
1117 - s : dot matches newlines
1118 */
1119 $pattern = '/(<([a-z0-9]+)[^>]*class="[^"]*\b' . $quotedClassName . '\b[^"]*"[^>]*>)(.*?)(<\/\2>)/is';
1120
1121 $currentIndex = 0;
1122 $result = preg_replace_callback($pattern, function($matches) use ($newContent, $targetIndex, &$currentIndex) {
1123 if ($currentIndex == $targetIndex) {
1124 $currentIndex++;
1125 return $matches[1] . $newContent . $matches[4];
1126 }
1127 $currentIndex++;
1128 return $matches[0]; // Return original match unchanged
1129 }, $html);
1130
1131 return $result;
1132 }
1133
1134 /**
1135 * Clean block name by removing namespace/plugin prefix
1136 *
1137 * @param string $block_name The full block name
1138 *
1139 * @return string Cleaned block name without prefix
1140 */
1141 public static function cleanBlockName( $block_name ) {
1142 // Remove namespace/plugin prefix (everything before the last slash)
1143 $parts = explode( '/', $block_name );
1144
1145 return end( $parts );
1146 }
1147
1148 /**
1149 * Escape invalid entities in HTML to prevent DOMDocument warnings.
1150 *
1151 * @param string $html The HTML string to escape.
1152 * @return string The escaped HTML string.
1153 */
1154 public static function escapeInvalidEntities($html) {
1155 // Replace & not followed by one of: #, a-z, A-Z, or 0-9, and then a semicolon
1156 return preg_replace('/&(?!(#[0-9]+|[a-zA-Z0-9]+);)/', '&amp;', $html);
1157 }
1158
1159 /**
1160 * Remove invalid blocks from array
1161 *
1162 * @param array $blocks Array of blocks to clean
1163 * @return array Cleaned array with only valid blocks
1164 */
1165 public static function cleanInvalidBlocks(array $blocks) {
1166 $cleanedBlocks = [];
1167
1168 foreach ($blocks as $block) {
1169 // Skip if not array
1170 if (!is_array($block)) {
1171 continue;
1172 }
1173
1174 // Skip if blockName is null or empty
1175 if (empty($block['blockName'])) {
1176 continue;
1177 }
1178
1179 // Skip if missing required properties
1180 if (!isset($block['attrs']) ||
1181 !isset($block['innerBlocks']) ||
1182 !isset($block['innerHTML']) ||
1183 !isset($block['innerContent'])) {
1184 continue;
1185 }
1186
1187 // Clean nested blocks recursively
1188 if (!empty($block['innerBlocks']) && is_array($block['innerBlocks'])) {
1189 $block['innerBlocks'] = self::cleanInvalidBlocks($block['innerBlocks']);
1190 }
1191
1192 $cleanedBlocks[] = $block;
1193 }
1194
1195 return $cleanedBlocks;
1196 }
1197 }
1198