| 1 |
<?php |
| 2 |
|
| 3 |
namespace Templately\Modules\AiContentMerger\Utils; |
| 4 |
|
| 5 |
/** |
| 6 |
* AI Content Helper |
| 7 |
* |
| 8 |
* Standalone, self-contained AI content merger (spec 022 FR-011): platform-aware merging |
| 9 |
* of AI-generated content into the original template — flatten-by-id, unicode |
| 10 |
* normalization, Elementor/Gutenberg dot-notation merge, and HTML class-based replacement. |
| 11 |
* |
| 12 |
* Every method is a pure static with ZERO cross-module dependencies. Session/path |
| 13 |
* resolution and file IO (reading the .ai.json / original JSON, writing the .ao.json debug |
| 14 |
* file) are the CONSUMER's responsibility (FSI's Finalizer). The public entry point is the |
| 15 |
* static merge() dispatcher. |
| 16 |
*/ |
| 17 |
class AIContentHelper { |
| 18 |
/** |
| 19 |
* Parity warnings accumulated during the most recent Gutenberg merge. |
| 20 |
* |
| 21 |
* A Gutenberg static block is validated by re-running its JS `save(attributes)` and |
| 22 |
* comparing the output against the markup stored in the post. So a merge that writes new |
| 23 |
* text into ONE of {attributes, markup} without the other produces a block the editor |
| 24 |
* reports as "Block contains unexpected or invalid content" — and, because the stored |
| 25 |
* markup is what the front end serves, a page that never shows the AI copy at all. |
| 26 |
* |
| 27 |
* Every replacement path below therefore moves both sides together. When one side cannot |
| 28 |
* be updated (the old string is not found where it was expected — entity/encoding drift, |
| 29 |
* an unknown block shape), that is recorded here instead of failing silently, so the |
| 30 |
* Finalizer can log it and the post-import rebuild pass knows which posts to visit. |
| 31 |
* |
| 32 |
* @var array<int,array{blockName:string,blockId:string,attribute:string,reason:string}> |
| 33 |
*/ |
| 34 |
private static $parity_warnings = []; |
| 35 |
|
| 36 |
public $htmlSources = [ |
| 37 |
'testimonial', |
| 38 |
'feature-list', |
| 39 |
'notice', |
| 40 |
'pricing-table', |
| 41 |
'typing-text', |
| 42 |
'interactive-promo', |
| 43 |
'call-to-action' |
| 44 |
]; |
| 45 |
|
| 46 |
/** |
| 47 |
* Merge AI-generated content into the original template for a given platform. |
| 48 |
* |
| 49 |
* Pure, self-contained dispatcher (spec 022 FR-011): the caller supplies the already-read |
| 50 |
* AI payload and original template; this routes to the platform-specific merger. Session |
| 51 |
* resolution and file IO are the consumer's responsibility, not this class's. |
| 52 |
* |
| 53 |
* @param array $payload The decoded AI-generated template content. |
| 54 |
* @param mixed $template The decoded original template to merge into. |
| 55 |
* @param string $platform Either 'elementor' or 'gutenberg'. |
| 56 |
* @return mixed The merged template, or the untouched $template for an unknown platform. |
| 57 |
*/ |
| 58 |
public static function merge( array $payload, $template, string $platform ) { |
| 59 |
if ( 'elementor' === $platform ) { |
| 60 |
return self::mergeAiContentWithOriginal( $payload, $template ); |
| 61 |
} |
| 62 |
if ( 'gutenberg' === $platform ) { |
| 63 |
return self::mergeAiContentWithOriginalGutenberg( $payload, $template ); |
| 64 |
} |
| 65 |
return $template; |
| 66 |
} |
| 67 |
|
| 68 |
/** |
| 69 |
* Normalize escape characters in AI content |
| 70 |
* |
| 71 |
* Removes backslashes before closing tags (converts `<\/` and `<\\/` to `</`) |
| 72 |
* This normalization is applied to all content values in the AI template's contents arrays. |
| 73 |
* |
| 74 |
* @param array $flat Reference to the flattened AI content array |
| 75 |
* @return void Modifies the array in place |
| 76 |
*/ |
| 77 |
public static function normalizeAiContentEscapeCharacters(&$flat) { |
| 78 |
foreach ($flat as &$element) { |
| 79 |
if (isset($element['contents']) && is_array($element['contents'])) { |
| 80 |
foreach ($element['contents'] as $index => &$content_item) { |
| 81 |
if (isset($content_item['content']) && is_string($content_item['content'])) { |
| 82 |
// Normalize content: unescape closing tags (remove backslash before </) |
| 83 |
$content_item['content'] = str_replace(['<\/', '<\\/'], '</', $content_item['content']); |
| 84 |
} |
| 85 |
else { |
| 86 |
unset($element['contents'][$index]); |
| 87 |
} |
| 88 |
} |
| 89 |
} |
| 90 |
} |
| 91 |
} |
| 92 |
|
| 93 |
/** |
| 94 |
* Recursively flatten a nested array by extracting elements with 'contents' and using their ID as key |
| 95 |
* |
| 96 |
* @param array $array The array to flatten |
| 97 |
* @param array $flat Reference to the flattened array |
| 98 |
* @return array The flattened array |
| 99 |
*/ |
| 100 |
public static function flattenById($array, &$flat = []) { |
| 101 |
foreach ($array as $key => $value) { |
| 102 |
if (is_array($value)) { |
| 103 |
// If this is an element with 'widgetType' and 'contents', use its parent key as ID |
| 104 |
if (isset($value['widgetType']) && isset($value['contents'])) { |
| 105 |
$flat[$key] = $value; |
| 106 |
} |
| 107 |
// Recurse into children |
| 108 |
self::flattenById($value, $flat); |
| 109 |
} |
| 110 |
} |
| 111 |
return $flat; |
| 112 |
} |
| 113 |
|
| 114 |
|
| 115 |
|
| 116 |
/** |
| 117 |
* Set a value in a nested array using a dot notation path |
| 118 |
* |
| 119 |
* @param array $array Reference to the array to modify |
| 120 |
* @param array $path The path as an array of keys |
| 121 |
* @param mixed $value The value to set |
| 122 |
*/ |
| 123 |
public static function setNestedValue(&$array, $path, $value) { |
| 124 |
$key = array_shift($path); |
| 125 |
|
| 126 |
if (empty($path)) { |
| 127 |
// We've reached the final key, set the value |
| 128 |
$array[$key] = $value; |
| 129 |
} else { |
| 130 |
// Initialize the nested array if it doesn't exist |
| 131 |
if (!isset($array[$key]) || !is_array($array[$key])) { |
| 132 |
$array[$key] = []; |
| 133 |
} |
| 134 |
|
| 135 |
// Continue recursively |
| 136 |
self::setNestedValue($array[$key], $path, $value); |
| 137 |
} |
| 138 |
} |
| 139 |
|
| 140 |
/** |
| 141 |
* Merge AI content with the original template |
| 142 |
* |
| 143 |
* @param array $ai_template_json The AI template JSON |
| 144 |
* @param array $original_template_json The original template JSON data |
| 145 |
* @return array The merged template JSON |
| 146 |
*/ |
| 147 |
public static function mergeAiContentWithOriginal($ai_template_json, $original_template_json) { |
| 148 |
// Some callers pass the AI template as a JSON-encoded string; decode it first. |
| 149 |
if (is_string($ai_template_json)) { |
| 150 |
$ai_template_json = json_decode($ai_template_json, true); |
| 151 |
} |
| 152 |
|
| 153 |
// 1. Flatten the AI template |
| 154 |
$flat = self::flattenById($ai_template_json); |
| 155 |
|
| 156 |
// 2. Normalize escape characters in AI content early in the pipeline |
| 157 |
self::normalizeAiContentEscapeCharacters($flat); |
| 158 |
|
| 159 |
$keys = array_keys($flat); |
| 160 |
|
| 161 |
// 3. Loop through original content only once and update elements directly |
| 162 |
self::updateElementorContentRecursively($flat, $keys, $original_template_json['content']); |
| 163 |
|
| 164 |
return $original_template_json; |
| 165 |
} |
| 166 |
|
| 167 |
/** |
| 168 |
* Update Elementor content recursively by looping through original content only once |
| 169 |
* |
| 170 |
* @param array $flat The flattened AI content array |
| 171 |
* @param array $keys Array of element IDs from the flat array |
| 172 |
* @param array $content Reference to the original content to update |
| 173 |
*/ |
| 174 |
public static function updateElementorContentRecursively($flat, array $keys, &$content) { |
| 175 |
if (!is_array($content)) { |
| 176 |
return; |
| 177 |
} |
| 178 |
|
| 179 |
// Check if this element has an ID and needs updating |
| 180 |
if (isset($content['id']) && in_array($content['id'], $keys)) { |
| 181 |
$element_id = $content['id']; |
| 182 |
$element = $flat[$element_id]; |
| 183 |
|
| 184 |
if (isset($element['contents'])) { |
| 185 |
// Update settings based on contents |
| 186 |
foreach ($element['contents'] as $item) { |
| 187 |
if (isset($item['attribute'], $item['content'])) { |
| 188 |
$content_value = is_string($item['content']) |
| 189 |
? str_replace(['<\/', '<\\/'], '</', $item['content']) |
| 190 |
: $item['content']; |
| 191 |
|
| 192 |
// Support for dot notation in attribute paths |
| 193 |
if (strpos($item['attribute'], '.') !== false) { |
| 194 |
$path = explode('.', $item['attribute']); |
| 195 |
self::setNestedValue($content['settings'], $path, $content_value); |
| 196 |
} else { |
| 197 |
$content['settings'][$item['attribute']] = $content_value; |
| 198 |
} |
| 199 |
} |
| 200 |
} |
| 201 |
} |
| 202 |
} |
| 203 |
|
| 204 |
// Recurse through elements array |
| 205 |
if (isset($content['elements']) && is_array($content['elements'])) { |
| 206 |
foreach ($content['elements'] as &$element) { |
| 207 |
self::updateElementorContentRecursively($flat, $keys, $element); |
| 208 |
} |
| 209 |
} |
| 210 |
|
| 211 |
// Recurse through all other array elements |
| 212 |
foreach ($content as &$value) { |
| 213 |
if (is_array($value)) { |
| 214 |
self::updateElementorContentRecursively($flat, $keys, $value); |
| 215 |
} |
| 216 |
} |
| 217 |
} |
| 218 |
|
| 219 |
/** |
| 220 |
* Merge AI content with the original Gutenberg template using advanced content replacement |
| 221 |
* |
| 222 |
* @param array $ai_template_json The AI template JSON |
| 223 |
* @param array $original_template_json The original template JSON data |
| 224 |
* @return array The merged template JSON |
| 225 |
*/ |
| 226 |
public static function mergeAiContentWithOriginalGutenberg($ai_template_json, $original_template_json) { |
| 227 |
if (empty($original_template_json['content'])) { |
| 228 |
return $original_template_json; |
| 229 |
} |
| 230 |
|
| 231 |
// Some callers pass the AI template as a JSON-encoded string; decode it first. |
| 232 |
if (is_string($ai_template_json)) { |
| 233 |
$ai_template_json = json_decode($ai_template_json, true); |
| 234 |
} |
| 235 |
|
| 236 |
self::$parity_warnings = []; |
| 237 |
|
| 238 |
// 1. Flatten the AI template by block ID (for blocks with 'contents') |
| 239 |
$flat = []; |
| 240 |
self::flattenGutenbergById($ai_template_json, $flat); |
| 241 |
|
| 242 |
// 2. Normalize escape characters in AI content early in the pipeline |
| 243 |
self::normalizeAiContentEscapeCharacters($flat); |
| 244 |
|
| 245 |
$generated = $flat; |
| 246 |
$keys = array_keys($generated); |
| 247 |
|
| 248 |
// 3. Parse the original Gutenberg content |
| 249 |
$blocks = parse_blocks($original_template_json['content']); |
| 250 |
|
| 251 |
// 4. Clean invalid blocks from parsed content |
| 252 |
$blocks = self::cleanInvalidBlocks($blocks); |
| 253 |
|
| 254 |
// 5. Replace content recursively using advanced replacer logic |
| 255 |
$blocks = self::replaceGutenbergContentRecursively($generated, $keys, $blocks); |
| 256 |
|
| 257 |
// 6. Clean invalid blocks before serialization |
| 258 |
$blocks = self::cleanInvalidBlocks($blocks); |
| 259 |
|
| 260 |
// 7. Serialize the updated blocks back to content |
| 261 |
$original_template_json['content'] = serialize_blocks($blocks); |
| 262 |
|
| 263 |
return $original_template_json; |
| 264 |
} |
| 265 |
|
| 266 |
/** |
| 267 |
* Replace content recursively in Gutenberg blocks (ported from GutenbergContentReplacer) |
| 268 |
*/ |
| 269 |
public static function replaceGutenbergContentRecursively($generated, array $keys, &$blocks) { |
| 270 |
$htmlSources = [ |
| 271 |
'testimonial', |
| 272 |
'feature-list', |
| 273 |
'notice', |
| 274 |
'pricing-table', |
| 275 |
'typing-text', |
| 276 |
'interactive-promo', |
| 277 |
'call-to-action' |
| 278 |
]; |
| 279 |
|
| 280 |
foreach ($blocks as &$block) { |
| 281 |
if (!empty($block['attrs']['blockId'])) { |
| 282 |
$blockId = $block['attrs']['blockId']; |
| 283 |
if (in_array($blockId, $keys)) { |
| 284 |
$blockData = $generated[$blockId]; |
| 285 |
$block_name = self::cleanBlockName( $block['blockName'] ); |
| 286 |
|
| 287 |
// Store old content BEFORE updating attributes |
| 288 |
$oldContentMap = []; |
| 289 |
if (!empty($blockData['contents']) && !in_array($block_name, $htmlSources)) { |
| 290 |
foreach ($blockData['contents'] as $content) { |
| 291 |
$attribute = $content['attribute']; |
| 292 |
$oldContent = self::getNestedGutenbergAttribute($block['attrs'], $attribute); |
| 293 |
if ($oldContent !== null) { |
| 294 |
$oldContentMap[$attribute] = $oldContent; |
| 295 |
} |
| 296 |
} |
| 297 |
} |
| 298 |
|
| 299 |
// Replace content in attributes |
| 300 |
if (!empty($blockData['contents']) && !in_array($block_name, $htmlSources)) { |
| 301 |
foreach ($blockData['contents'] as $content) { |
| 302 |
$attribute = $content['attribute']; |
| 303 |
$newContent = $content['content']; |
| 304 |
self::setNestedGutenbergAttribute($block['attrs'], $attribute, $newContent); |
| 305 |
} |
| 306 |
} |
| 307 |
|
| 308 |
// Replace content in innerHTML and innerContent using old content |
| 309 |
if (!empty($block['innerHTML']) || !empty($block['innerContent'])) { |
| 310 |
self::replaceInGutenbergHtmlContent($block, $blockData, $oldContentMap); |
| 311 |
} |
| 312 |
|
| 313 |
if(in_array($block_name, $htmlSources)){ |
| 314 |
if (!empty($blockData['contents'])) { |
| 315 |
// These blocks are addressed by CLASS NAME because that is where their copy |
| 316 |
// actually lives: every block in $htmlSources declares its text attributes |
| 317 |
// with a `source` + `selector` (EB's notice -> source:"text" on |
| 318 |
// `.eb-notice-title`; feature-list -> source:"query" over the `li`s), so the |
| 319 |
// values are PARSED BACK OUT OF THE MARKUP and never stored in the block |
| 320 |
// comment. Rewriting the markup is therefore the complete fix for them. |
| 321 |
// |
| 322 |
// The attribute pass below is a safety net for the mixed case — a block that |
| 323 |
// ALSO keeps a comment-stored copy of the same string, which would then go |
| 324 |
// stale. Read the old text out of the markup FIRST: after the replacement it |
| 325 |
// is gone, and it is the only key that can identify such an attribute (there |
| 326 |
// is no class-name -> attribute map, and inventing one would rot on every EB |
| 327 |
// markup change). |
| 328 |
$oldByClass = self::extractContentByClassName( |
| 329 |
is_string($block['innerHTML'] ?? null) ? $block['innerHTML'] : '', |
| 330 |
$blockData['contents'] |
| 331 |
); |
| 332 |
|
| 333 |
if (!empty($block['innerHTML'])) { |
| 334 |
$block['innerHTML'] = self::replaceContentByClassName($block['innerHTML'], $blockData['contents']); |
| 335 |
} |
| 336 |
if (!empty($block['innerContent']) && is_array($block['innerContent'])) { |
| 337 |
foreach ($block['innerContent'] as &$content) { |
| 338 |
if (is_string($content)) { |
| 339 |
$content = self::replaceContentByClassName($content, $blockData['contents']); |
| 340 |
} |
| 341 |
} |
| 342 |
unset($content); |
| 343 |
} |
| 344 |
|
| 345 |
self::syncAttrsFromClassReplacements($block, $blockData['contents'], $oldByClass); |
| 346 |
} |
| 347 |
} |
| 348 |
if($block["blockName"] === 'essential-blocks/accordion'){ |
| 349 |
$block_inner_block_ids = array_map(function($innerBlock) { |
| 350 |
return $innerBlock["attrs"]["blockId"] ?? null; |
| 351 |
}, $block["innerBlocks"]); |
| 352 |
$_generated = array_fill_keys($block_inner_block_ids, ['contents' => $blockData['contents']]); |
| 353 |
|
| 354 |
if(isset($block["innerBlocks"][0]["attrs"]["accordionLists"]) && count($block["innerBlocks"][0]["attrs"]["accordionLists"]) > 1){ |
| 355 |
$block["innerBlocks"] = self::replaceGutenbergContentRecursively($_generated, $block_inner_block_ids, $block['innerBlocks']); |
| 356 |
} |
| 357 |
else { |
| 358 |
// The live path: EB syncs each accordion-item's `accordionLists` down from |
| 359 |
// the parent as a SINGLE-entry array (accordion-item/src/edit.js), so the |
| 360 |
// count > 1 branch above never fires for current EB markup. |
| 361 |
// |
| 362 |
// `accordion-item/src/save.js` renders `foundItem?.title` straight out of |
| 363 |
// this attribute, so writing the updated entry here WITHOUT rewriting the |
| 364 |
// item's own markup guarantees a save()/markup mismatch: the editor shows |
| 365 |
// "Attempt recovery" on every FAQ row and the front end keeps serving the |
| 366 |
// pre-AI titles. Move both sides together. |
| 367 |
$attrAccordionLists = $block["attrs"]["accordionLists"]; |
| 368 |
$ids = array_column($attrAccordionLists, 'id'); |
| 369 |
|
| 370 |
foreach ($block["innerBlocks"] as $key => $accordion) { |
| 371 |
$itemLists = $accordion["attrs"]["accordionLists"] ?? []; |
| 372 |
if (!is_array($itemLists)) { |
| 373 |
continue; |
| 374 |
} |
| 375 |
|
| 376 |
foreach($itemLists as $accordionKey => $accordionList){ |
| 377 |
if (!is_array($accordionList) || !isset($accordionList['id'])) { |
| 378 |
continue; |
| 379 |
} |
| 380 |
|
| 381 |
// search $block["attrs"]["accordionLists"] by $accordionList["id"] and replace $accordionList with searched one |
| 382 |
$foundIndex = array_search($accordionList["id"], $ids); |
| 383 |
if ($foundIndex === false) { |
| 384 |
continue; |
| 385 |
} |
| 386 |
|
| 387 |
$updatedEntry = $attrAccordionLists[$foundIndex]; |
| 388 |
$pairs = self::diffStringFields($accordionList, $updatedEntry); |
| 389 |
|
| 390 |
$block["innerBlocks"][$key]["attrs"]["accordionLists"][$accordionKey] = $updatedEntry; |
| 391 |
|
| 392 |
if (!empty($pairs)) { |
| 393 |
self::applyReplacementsToBlockMarkup( |
| 394 |
$block["innerBlocks"][$key], |
| 395 |
self::buildReplacementRecords($pairs) |
| 396 |
); |
| 397 |
} |
| 398 |
} |
| 399 |
} |
| 400 |
} |
| 401 |
} |
| 402 |
} |
| 403 |
|
| 404 |
// Process nested blocks recursively |
| 405 |
if (!empty($block['innerBlocks'])) { |
| 406 |
$block['innerBlocks'] = self::replaceGutenbergContentRecursively($generated, $keys, $block['innerBlocks']); |
| 407 |
} |
| 408 |
} |
| 409 |
} |
| 410 |
return $blocks; |
| 411 |
} |
| 412 |
|
| 413 |
/** |
| 414 |
* Parity warnings from the most recent Gutenberg merge. |
| 415 |
* |
| 416 |
* Empty means every replacement landed on both the attributes and the markup. A non-empty |
| 417 |
* list names the blocks where one side could not be updated — those are exactly the blocks |
| 418 |
* that will open with a recovery banner, so consumers (the FSI Finalizer log, the |
| 419 |
* post-import rebuild queue) use it to decide what to report and what to revisit. |
| 420 |
* |
| 421 |
* @return array<int,array{blockName:string,blockId:string,attribute:string,reason:string}> |
| 422 |
*/ |
| 423 |
public static function collect_parity_warnings(): array { |
| 424 |
return self::$parity_warnings; |
| 425 |
} |
| 426 |
|
| 427 |
/** |
| 428 |
* Record one attribute/markup parity failure. Never throws — a merge that cannot keep the |
| 429 |
* two sides in step must still produce content. |
| 430 |
*/ |
| 431 |
private static function addParityWarning($blockName, $blockId, $attribute, $reason) { |
| 432 |
self::$parity_warnings[] = [ |
| 433 |
'blockName' => (string) $blockName, |
| 434 |
'blockId' => (string) $blockId, |
| 435 |
'attribute' => (string) $attribute, |
| 436 |
'reason' => (string) $reason, |
| 437 |
]; |
| 438 |
} |
| 439 |
|
| 440 |
/** |
| 441 |
* String fields that differ between two versions of the same structured entry. |
| 442 |
* |
| 443 |
* Deliberately NOT limited to `title`: an accordion entry also carries |
| 444 |
* `titlePrefixText`/`titleSuffixText`/`imageAlt`, and EB adds fields over time. Comparing |
| 445 |
* every string field keeps this correct as the shape grows instead of silently covering |
| 446 |
* one key. |
| 447 |
* |
| 448 |
* @param array $old Entry before the update. |
| 449 |
* @param array $new Entry after the update. |
| 450 |
* @return array<int,array{attribute:string,old:string,new:string}> |
| 451 |
*/ |
| 452 |
private static function diffStringFields($old, $new) { |
| 453 |
$pairs = []; |
| 454 |
|
| 455 |
if (!is_array($old) || !is_array($new)) { |
| 456 |
return $pairs; |
| 457 |
} |
| 458 |
|
| 459 |
foreach ($new as $field => $value) { |
| 460 |
if (!is_string($value) || !isset($old[$field]) || !is_string($old[$field])) { |
| 461 |
continue; |
| 462 |
} |
| 463 |
if ($old[$field] === '' || $old[$field] === $value) { |
| 464 |
continue; |
| 465 |
} |
| 466 |
$pairs[] = [ |
| 467 |
'attribute' => (string) $field, |
| 468 |
'old' => $old[$field], |
| 469 |
'new' => $value, |
| 470 |
]; |
| 471 |
} |
| 472 |
|
| 473 |
return $pairs; |
| 474 |
} |
| 475 |
|
| 476 |
/** |
| 477 |
* Shape old/new pairs into the replacement records `replaceGutenbergContentInHtml()` expects. |
| 478 |
* |
| 479 |
* Longest-first, for the same reason `replaceInGutenbergHtmlContent()` sorts: a short string |
| 480 |
* that is a prefix of a longer one would otherwise consume it. |
| 481 |
* |
| 482 |
* @param array<int,array{attribute:string,old:string,new:string}> $pairs |
| 483 |
* @return array |
| 484 |
*/ |
| 485 |
private static function buildReplacementRecords($pairs) { |
| 486 |
$replacements = []; |
| 487 |
|
| 488 |
foreach ($pairs as $pair) { |
| 489 |
$replacements[] = [ |
| 490 |
'originalFormat' => $pair['old'], |
| 491 |
'decodedFormat' => json_decode('"' . $pair['old'] . '"'), |
| 492 |
'normalizedFormat' => self::normalizeGutenbergUnicodeContent($pair['old']), |
| 493 |
'newContent' => $pair['new'], |
| 494 |
'attribute' => $pair['attribute'], |
| 495 |
]; |
| 496 |
} |
| 497 |
|
| 498 |
usort($replacements, function ($a, $b) { |
| 499 |
return strlen($b['originalFormat']) - strlen($a['originalFormat']); |
| 500 |
}); |
| 501 |
|
| 502 |
return $replacements; |
| 503 |
} |
| 504 |
|
| 505 |
/** |
| 506 |
* Apply replacement records to a block's own markup (`innerHTML` + `innerContent`). |
| 507 |
* |
| 508 |
* Records a parity warning for any replacement whose old text is still present afterwards — |
| 509 |
* that block's save() output will not match what is stored. |
| 510 |
* |
| 511 |
* @param array $block Block, by reference. |
| 512 |
* @param array $replacements Records from {@see self::buildReplacementRecords()}. |
| 513 |
*/ |
| 514 |
private static function applyReplacementsToBlockMarkup(&$block, $replacements) { |
| 515 |
if (empty($replacements)) { |
| 516 |
return; |
| 517 |
} |
| 518 |
|
| 519 |
$before = self::ownMarkup($block); |
| 520 |
|
| 521 |
if (!empty($block['innerHTML']) && is_string($block['innerHTML'])) { |
| 522 |
$block['innerHTML'] = self::replaceGutenbergContentInHtml($block['innerHTML'], $replacements); |
| 523 |
} |
| 524 |
|
| 525 |
if (!empty($block['innerContent']) && is_array($block['innerContent'])) { |
| 526 |
foreach ($block['innerContent'] as $index => $chunk) { |
| 527 |
if (is_string($chunk)) { |
| 528 |
$block['innerContent'][$index] = self::replaceGutenbergContentInHtml($chunk, $replacements); |
| 529 |
} |
| 530 |
} |
| 531 |
} |
| 532 |
|
| 533 |
$after = self::ownMarkup($block); |
| 534 |
$hasChildren = !empty($block['innerBlocks']); |
| 535 |
|
| 536 |
foreach ($replacements as $replacement) { |
| 537 |
$old = $replacement['originalFormat']; |
| 538 |
$new = $replacement['newContent']; |
| 539 |
|
| 540 |
if ($old === '' || $new === '') { |
| 541 |
continue; |
| 542 |
} |
| 543 |
|
| 544 |
$wasHere = strpos($before, $old) !== false; |
| 545 |
|
| 546 |
if ($wasHere) { |
| 547 |
// The copy IS in this block's markup. If it survived the replacement the block is |
| 548 |
// now desynced from its own attributes — this is the "Attempt recovery" case. |
| 549 |
if (strpos($after, $old) !== false) { |
| 550 |
self::addParityWarning($block['blockName'] ?? '', $block['attrs']['blockId'] ?? '', $replacement['attribute'], 'markup_not_updated'); |
| 551 |
} |
| 552 |
continue; |
| 553 |
} |
| 554 |
|
| 555 |
// The copy was not in this block's own markup. `parse_blocks` gives each block only |
| 556 |
// its OWN chunks, so for a container (an accordion parent, a wrapper) the text |
| 557 |
// legitimately lives in a child block that is walked separately — silence there. |
| 558 |
// With no children there is nowhere else for it to be, so the replacement had no |
| 559 |
// target and the two sides cannot agree. |
| 560 |
if (!$hasChildren && strpos($after, $new) === false && trim($after) !== '') { |
| 561 |
self::addParityWarning($block['blockName'] ?? '', $block['attrs']['blockId'] ?? '', $replacement['attribute'], 'markup_not_updated'); |
| 562 |
} |
| 563 |
} |
| 564 |
} |
| 565 |
|
| 566 |
/** |
| 567 |
* A block's OWN markup — `innerHTML` plus its string `innerContent` chunks, never its |
| 568 |
* children's (`parse_blocks` keeps those in `innerBlocks`, walked separately). |
| 569 |
* |
| 570 |
* @param array $block |
| 571 |
* @return string |
| 572 |
*/ |
| 573 |
private static function ownMarkup($block) { |
| 574 |
$markup = is_string($block['innerHTML'] ?? null) ? $block['innerHTML'] : ''; |
| 575 |
|
| 576 |
if (!empty($block['innerContent']) && is_array($block['innerContent'])) { |
| 577 |
foreach ($block['innerContent'] as $chunk) { |
| 578 |
if (is_string($chunk)) { |
| 579 |
$markup .= $chunk; |
| 580 |
} |
| 581 |
} |
| 582 |
} |
| 583 |
|
| 584 |
return $markup; |
| 585 |
} |
| 586 |
|
| 587 |
/** |
| 588 |
* Current text of each class-addressed target, keyed by the class name used to address it. |
| 589 |
* |
| 590 |
* Must be called BEFORE the class-based replacement runs — afterwards the old text is gone, |
| 591 |
* and it is the only thing that can identify the attribute holding the same copy. |
| 592 |
* |
| 593 |
* @param string $html Block markup. |
| 594 |
* @param array $contents `[['attribute' => className, 'content' => newContent], …]` |
| 595 |
* @return array<string,string> className => old text |
| 596 |
*/ |
| 597 |
public static function extractContentByClassName($html, $contents) { |
| 598 |
$found = []; |
| 599 |
|
| 600 |
if (!is_string($html) || $html === '' || empty($contents)) { |
| 601 |
return $found; |
| 602 |
} |
| 603 |
if (!class_exists('DOMDocument') || !class_exists('DOMXPath')) { |
| 604 |
return $found; |
| 605 |
} |
| 606 |
|
| 607 |
$dom = new \DOMDocument(); |
| 608 |
@$dom->loadHTML('<?xml encoding="utf-8" ?>' . self::escapeInvalidEntities($html), LIBXML_HTML_NOIMPLIED | LIBXML_HTML_NODEFDTD); |
| 609 |
$xpath = new \DOMXPath($dom); |
| 610 |
|
| 611 |
foreach ($contents as $item) { |
| 612 |
if (!isset($item['attribute'])) { |
| 613 |
continue; |
| 614 |
} |
| 615 |
$className = $item['attribute']; |
| 616 |
$baseClassName = self::extractBaseClassName($className); |
| 617 |
$targetIndex = self::extractClassIndex($className); |
| 618 |
|
| 619 |
$nodes = $xpath->query("//*[contains(concat(' ', normalize-space(@class), ' '), ' $baseClassName ')]"); |
| 620 |
if (!$nodes || $nodes->length === 0) { |
| 621 |
continue; |
| 622 |
} |
| 623 |
|
| 624 |
$node = $targetIndex !== null ? ($nodes[$targetIndex] ?? null) : $nodes[0]; |
| 625 |
if ($node !== null) { |
| 626 |
$found[$className] = $node->nodeValue; |
| 627 |
} |
| 628 |
} |
| 629 |
|
| 630 |
return $found; |
| 631 |
} |
| 632 |
|
| 633 |
/** |
| 634 |
* Write class-replaced copy back into the block's attributes, matched BY VALUE. |
| 635 |
* |
| 636 |
* For the `$htmlSources` blocks the copy normally lives ONLY in the markup (their text |
| 637 |
* attributes are `source`d from selectors), so this usually finds nothing — that is the |
| 638 |
* expected, healthy case and is deliberately NOT warned about. It exists for the mixed |
| 639 |
* block that also keeps a comment-stored copy of the same string, which would otherwise go |
| 640 |
* stale and make save() disagree with the markup. |
| 641 |
* |
| 642 |
* Matching on the exact old string needs no class-name -> attribute table and cannot rot |
| 643 |
* when EB renames a class; a table would have to be maintained per block per release, and a |
| 644 |
* stale entry fails silently. |
| 645 |
* |
| 646 |
* Exact FULL-string equality only (never substring), and only on string scalars, so an |
| 647 |
* unrelated attribute that merely contains the text is left alone. |
| 648 |
* |
| 649 |
* @param array $block Block, by reference. |
| 650 |
* @param array $contents `[['attribute' => className, 'content' => newContent], …]` |
| 651 |
* @param array $oldByClass className => old text, from {@see self::extractContentByClassName()} |
| 652 |
*/ |
| 653 |
private static function syncAttrsFromClassReplacements(&$block, $contents, $oldByClass) { |
| 654 |
if (empty($contents) || empty($oldByClass) || empty($block['attrs']) || !is_array($block['attrs'])) { |
| 655 |
return; |
| 656 |
} |
| 657 |
|
| 658 |
$map = []; |
| 659 |
foreach ($contents as $item) { |
| 660 |
if (!isset($item['attribute'], $item['content']) || !is_string($item['content'])) { |
| 661 |
continue; |
| 662 |
} |
| 663 |
$className = $item['attribute']; |
| 664 |
if (!isset($oldByClass[$className])) { |
| 665 |
continue; |
| 666 |
} |
| 667 |
|
| 668 |
$old = $oldByClass[$className]; |
| 669 |
if (!is_string($old)) { |
| 670 |
continue; |
| 671 |
} |
| 672 |
|
| 673 |
$old = trim($old); |
| 674 |
if ($old === '' || $old === $item['content']) { |
| 675 |
continue; |
| 676 |
} |
| 677 |
|
| 678 |
$map[$old] = ['new' => $item['content'], 'attribute' => $className]; |
| 679 |
} |
| 680 |
|
| 681 |
if (empty($map)) { |
| 682 |
return; |
| 683 |
} |
| 684 |
|
| 685 |
$written = []; |
| 686 |
self::replaceStringsInAttrs($block['attrs'], $map, $written); |
| 687 |
} |
| 688 |
|
| 689 |
/** |
| 690 |
* Recursively replace exact string values inside an attribute tree. |
| 691 |
* |
| 692 |
* @param mixed $attrs Attribute value/tree, by reference. |
| 693 |
* @param array $map oldString => ['new' => newString, 'attribute' => label] |
| 694 |
* @param array $written oldString => hit count, by reference. |
| 695 |
*/ |
| 696 |
private static function replaceStringsInAttrs(&$attrs, $map, &$written) { |
| 697 |
if (is_string($attrs)) { |
| 698 |
$key = trim($attrs); |
| 699 |
if (isset($map[$key])) { |
| 700 |
$attrs = $map[$key]['new']; |
| 701 |
$written[$key] = ($written[$key] ?? 0) + 1; |
| 702 |
} |
| 703 |
return; |
| 704 |
} |
| 705 |
|
| 706 |
if (is_array($attrs)) { |
| 707 |
foreach ($attrs as $key => $value) { |
| 708 |
self::replaceStringsInAttrs($attrs[$key], $map, $written); |
| 709 |
} |
| 710 |
} |
| 711 |
} |
| 712 |
|
| 713 |
/** |
| 714 |
* Set nested attribute value using dot notation (ported from GutenbergContentReplacer) |
| 715 |
*/ |
| 716 |
public static function setNestedGutenbergAttribute(&$attrs, $path, $value) { |
| 717 |
$keys = explode('.', $path); |
| 718 |
$current = &$attrs; |
| 719 |
for ($i = 0; $i < count($keys) - 1; $i++) { |
| 720 |
$key = $keys[$i]; |
| 721 |
if (!isset($current[$key])) { |
| 722 |
$current[$key] = []; |
| 723 |
} |
| 724 |
$current = &$current[$key]; |
| 725 |
} |
| 726 |
$finalKey = end($keys); |
| 727 |
$current[$finalKey] = $value; |
| 728 |
} |
| 729 |
|
| 730 |
/** |
| 731 |
* Get nested attribute value using dot notation (ported from GutenbergContentReplacer) |
| 732 |
*/ |
| 733 |
public static function getNestedGutenbergAttribute($attrs, $path) { |
| 734 |
$keys = explode('.', $path); |
| 735 |
$current = $attrs; |
| 736 |
foreach ($keys as $key) { |
| 737 |
if (!isset($current[$key])) { |
| 738 |
return null; |
| 739 |
} |
| 740 |
$current = $current[$key]; |
| 741 |
} |
| 742 |
return $current; |
| 743 |
} |
| 744 |
|
| 745 |
/** |
| 746 |
* Replace content in innerHTML and innerContent while preserving HTML structure (ported from GutenbergContentReplacer) |
| 747 |
*/ |
| 748 |
public static function replaceInGutenbergHtmlContent(&$block, $blockData, $oldContentMap) { |
| 749 |
if (empty($blockData['contents']) || empty($oldContentMap)) return; |
| 750 |
$replacements = []; |
| 751 |
foreach ($blockData['contents'] as $content) { |
| 752 |
$attribute = $content['attribute']; |
| 753 |
$newContent = $content['content']; |
| 754 |
if (isset($oldContentMap[$attribute])) { |
| 755 |
$oldAttributeContent = $oldContentMap[$attribute]; |
| 756 |
$decodedUnicodeContent = json_decode('"' . $oldAttributeContent . '"'); |
| 757 |
$normalizedAttributeContent = self::normalizeGutenbergUnicodeContent($oldAttributeContent); |
| 758 |
$normalizedNewContent = self::normalizeGutenbergUnicodeContent($newContent); |
| 759 |
if ($normalizedAttributeContent !== $normalizedNewContent) { |
| 760 |
$replacements[] = [ |
| 761 |
'originalFormat' => $oldAttributeContent, |
| 762 |
'decodedFormat' => $decodedUnicodeContent, |
| 763 |
'normalizedFormat' => $normalizedAttributeContent, |
| 764 |
'newContent' => $newContent, |
| 765 |
'attribute' => $attribute |
| 766 |
]; |
| 767 |
} |
| 768 |
} |
| 769 |
} |
| 770 |
// sort $replacements by length of 'originalFormat' in descending order |
| 771 |
usort($replacements, function($a, $b) { |
| 772 |
return strlen($b['originalFormat']) - strlen($a['originalFormat']); |
| 773 |
}); |
| 774 |
if (empty($replacements)) { |
| 775 |
return; |
| 776 |
} |
| 777 |
|
| 778 |
// One call, so the leftover check below sees the SAME markup the block ends up with — |
| 779 |
// the attributes have already been rewritten by the caller, so anything this fails to |
| 780 |
// replace is a desync that will surface as "Attempt recovery" in the editor. |
| 781 |
self::applyReplacementsToBlockMarkup($block, $replacements); |
| 782 |
} |
| 783 |
|
| 784 |
/** |
| 785 |
* Replace content in HTML while preserving structure and handling Unicode (ported from GutenbergContentReplacer) |
| 786 |
* |
| 787 |
* Uses targeted replacement that avoids replacing text inside HTML attributes (href, src, data-*, etc.) |
| 788 |
* to prevent breaking URLs and other attribute values. |
| 789 |
*/ |
| 790 |
public static function replaceGutenbergContentInHtml($html, $replacements) { |
| 791 |
foreach ($replacements as $replacement) { |
| 792 |
$originalFormat = $replacement['originalFormat']; |
| 793 |
$decodedFormat = $replacement['decodedFormat']; |
| 794 |
$normalizedFormat = $replacement['normalizedFormat']; |
| 795 |
$newContent = $replacement['newContent']; |
| 796 |
if (empty($originalFormat)) continue; |
| 797 |
$htmlNewContent = $newContent; |
| 798 |
|
| 799 |
// Use targeted replacement that avoids HTML attributes |
| 800 |
$html = self::replaceTextOutsideAttributes($html, $originalFormat, $htmlNewContent); |
| 801 |
|
| 802 |
if ($decodedFormat !== null && $decodedFormat !== $originalFormat) { |
| 803 |
$html = self::replaceTextOutsideAttributes($html, $decodedFormat, $htmlNewContent); |
| 804 |
} |
| 805 |
if ($normalizedFormat !== $decodedFormat && $normalizedFormat !== $originalFormat) { |
| 806 |
$html = self::replaceTextOutsideAttributes($html, $normalizedFormat, $htmlNewContent); |
| 807 |
} |
| 808 |
} |
| 809 |
return $html; |
| 810 |
} |
| 811 |
|
| 812 |
/** |
| 813 |
* Replace text in HTML only outside of HTML tags and attributes |
| 814 |
* |
| 815 |
* This function replaces occurrences of $oldText with $newText, but only when the text |
| 816 |
* appears outside of HTML tags and attributes. This prevents unintended replacements |
| 817 |
* inside URLs, src attributes, href attributes, and other HTML attributes. |
| 818 |
* |
| 819 |
* @param string $html The HTML content to process |
| 820 |
* @param string $oldText The text to find and replace |
| 821 |
* @param string $newText The replacement text |
| 822 |
* @return string The HTML with replacements applied only outside of tags/attributes |
| 823 |
*/ |
| 824 |
private static function replaceTextOutsideAttributes($html, $oldText, $newText) { |
| 825 |
if (empty($oldText) || $oldText === $newText) { |
| 826 |
return $html; |
| 827 |
} |
| 828 |
|
| 829 |
$result = ''; |
| 830 |
$lastPos = 0; |
| 831 |
|
| 832 |
// Find all occurrences of the text |
| 833 |
while (($pos = strpos($html, $oldText, $lastPos)) !== false) { |
| 834 |
// Check if this occurrence is inside an HTML tag or attribute |
| 835 |
if (!self::isPositionInsideTag($html, $pos)) { |
| 836 |
// Not inside a tag, safe to replace |
| 837 |
$result .= substr($html, $lastPos, $pos - $lastPos) . $newText; |
| 838 |
$lastPos = $pos + strlen($oldText); |
| 839 |
} else { |
| 840 |
// Inside a tag, skip this occurrence |
| 841 |
$result .= substr($html, $lastPos, $pos - $lastPos + strlen($oldText)); |
| 842 |
$lastPos = $pos + strlen($oldText); |
| 843 |
} |
| 844 |
} |
| 845 |
|
| 846 |
// Append remaining HTML |
| 847 |
$result .= substr($html, $lastPos); |
| 848 |
return $result; |
| 849 |
} |
| 850 |
|
| 851 |
/** |
| 852 |
* Check if a position in HTML is inside an HTML attribute value |
| 853 |
* |
| 854 |
* This checks if the position is between quotes within an HTML tag. |
| 855 |
* Returns true only if the position is inside an attribute value (between quotes), |
| 856 |
* not just anywhere inside a tag. |
| 857 |
* |
| 858 |
* @param string $html The HTML content |
| 859 |
* @param int $position The position to check |
| 860 |
* @return bool True if the position is inside an attribute value, false otherwise |
| 861 |
*/ |
| 862 |
private static function isPositionInsideTag($html, $position) { |
| 863 |
// Get the text before the position |
| 864 |
$beforeText = substr($html, 0, $position); |
| 865 |
|
| 866 |
// Find the last < and > before the position |
| 867 |
$lastOpenTag = strrpos($beforeText, '<'); |
| 868 |
$lastCloseTag = strrpos($beforeText, '>'); |
| 869 |
|
| 870 |
// If there's no unclosed tag, we're not inside a tag |
| 871 |
if ($lastOpenTag === false || ($lastCloseTag !== false && $lastOpenTag < $lastCloseTag)) { |
| 872 |
return false; |
| 873 |
} |
| 874 |
|
| 875 |
// We're inside a tag. Now check if we're inside an attribute value (between quotes) |
| 876 |
// Get the tag content from the last < to the position |
| 877 |
$tagContent = substr($html, $lastOpenTag, $position - $lastOpenTag); |
| 878 |
|
| 879 |
// Track whether we're inside double or single quotes by iterating through the tag content |
| 880 |
$inDoubleQuotes = false; |
| 881 |
$inSingleQuotes = false; |
| 882 |
|
| 883 |
for ($i = 0; $i < strlen($tagContent); $i++) { |
| 884 |
$char = $tagContent[$i]; |
| 885 |
|
| 886 |
// Toggle quote state when we encounter a quote |
| 887 |
if ($char === '"' && !$inSingleQuotes) { |
| 888 |
$inDoubleQuotes = !$inDoubleQuotes; |
| 889 |
} elseif ($char === "'" && !$inDoubleQuotes) { |
| 890 |
$inSingleQuotes = !$inSingleQuotes; |
| 891 |
} |
| 892 |
} |
| 893 |
|
| 894 |
// We're inside an attribute value if we're inside either type of quotes |
| 895 |
return $inDoubleQuotes || $inSingleQuotes; |
| 896 |
} |
| 897 |
|
| 898 |
/** |
| 899 |
* Normalize Unicode content to handle different apostrophe types and other Unicode variations (ported from GutenbergContentReplacer) |
| 900 |
*/ |
| 901 |
public static function normalizeGutenbergUnicodeContent($content) { |
| 902 |
$decoded = json_decode('"' . $content . '"'); |
| 903 |
if ($decoded !== null) { |
| 904 |
$content = $decoded; |
| 905 |
} |
| 906 |
$unicodeReplacements = [ |
| 907 |
'\u2019' => "'", |
| 908 |
'\u2018' => "'", |
| 909 |
'\u201C' => '"', |
| 910 |
'\u201D' => '"', |
| 911 |
'\u2013' => '-', |
| 912 |
'\u2014' => '-', |
| 913 |
'\u2026' => '...', |
| 914 |
"\u{2019}" => "'", |
| 915 |
"\u{2018}" => "'", |
| 916 |
"\u{201C}" => '"', |
| 917 |
"\u{201D}" => '"', |
| 918 |
"\u{2013}" => '-', |
| 919 |
"\u{2014}" => '-', |
| 920 |
"\u{2026}" => '...' |
| 921 |
]; |
| 922 |
return str_replace(array_keys($unicodeReplacements), array_values($unicodeReplacements), $content); |
| 923 |
} |
| 924 |
|
| 925 |
/** |
| 926 |
* Convert content to HTML format (handle line breaks and inline tags) (ported from GutenbergContentReplacer) |
| 927 |
*/ |
| 928 |
public static function convertGutenbergToHtmlFormat($content) { |
| 929 |
$content = str_replace("\n", '<br>', $content); |
| 930 |
$content = str_replace("\r\n", '<br>', $content); |
| 931 |
return $content; |
| 932 |
} |
| 933 |
|
| 934 |
/** |
| 935 |
* Recursively flatten a nested Gutenberg AI array by extracting blocks with 'contents' and using their blockId as key |
| 936 |
* |
| 937 |
* @param array $array The array to flatten |
| 938 |
* @param array $flat Reference to the flattened array |
| 939 |
* @return array The flattened array |
| 940 |
*/ |
| 941 |
public static function flattenGutenbergById($array, &$flat = []) { |
| 942 |
foreach ($array as $key => $value) { |
| 943 |
if (is_array($value)) { |
| 944 |
// If this is a block with 'blockName' and 'contents', use its parent key as ID |
| 945 |
if (isset($value['blockName']) && isset($value['contents'])) { |
| 946 |
$flat[$key] = $value; |
| 947 |
} |
| 948 |
// Recurse into children |
| 949 |
self::flattenGutenbergById($value, $flat); |
| 950 |
} |
| 951 |
} |
| 952 |
return $flat; |
| 953 |
} |
| 954 |
|
| 955 |
/** |
| 956 |
* Replace the inner content of tags with given class names in the HTML. |
| 957 |
* Supports indexed class names (e.g., "eb-feature-list-title.0", "eb-feature-list-title.1"). |
| 958 |
* Falls back to regex if DOMDocument does not find the class. |
| 959 |
* |
| 960 |
* @param string $html The HTML string. |
| 961 |
* @param array $contents Array of ['attribute' => className, 'content' => newContent] |
| 962 |
* @return string The updated HTML. |
| 963 |
*/ |
| 964 |
public static function replaceContentByClassName($html, $contents) { |
| 965 |
$classExists = false; |
| 966 |
foreach ($contents as $item) { |
| 967 |
$className = $item['attribute']; |
| 968 |
// Extract base class name (remove index if present) |
| 969 |
$baseClassName = self::extractBaseClassName($className); |
| 970 |
if (preg_match('/class=["\'][^"\']*\b' . preg_quote($baseClassName, '/') . '\b[^"\']*["\']/', $html)) { |
| 971 |
$classExists = true; |
| 972 |
break; |
| 973 |
} |
| 974 |
} |
| 975 |
if (!$classExists) { |
| 976 |
return $html; // No relevant class found, skip both methods |
| 977 |
} |
| 978 |
|
| 979 |
if (class_exists('DOMDocument') && class_exists('DOMXPath')) { |
| 980 |
return self::replaceContentByClassNameDom($html, $contents); |
| 981 |
} else { |
| 982 |
return self::replaceContentByClassNameRegex($html, $contents); |
| 983 |
} |
| 984 |
} |
| 985 |
|
| 986 |
/** |
| 987 |
* Extract base class name from indexed class name. |
| 988 |
* |
| 989 |
* @param string $className The class name (e.g., "eb-feature-list-title.0") |
| 990 |
* @return string The base class name (e.g., "eb-feature-list-title") |
| 991 |
*/ |
| 992 |
public static function extractBaseClassName($className) { |
| 993 |
// Check if class name has numeric index at the end |
| 994 |
if (preg_match('/^(.+)\.(\d+)$/', $className, $matches)) { |
| 995 |
return $matches[1]; // Return base class name |
| 996 |
} |
| 997 |
return $className; // Return original if no index found |
| 998 |
} |
| 999 |
|
| 1000 |
/** |
| 1001 |
* Extract index from indexed class name. |
| 1002 |
* |
| 1003 |
* @param string $className The class name (e.g., "eb-feature-list-title.0") |
| 1004 |
* @return int|null The index (e.g., 0) or null if no index found |
| 1005 |
*/ |
| 1006 |
public static function extractClassIndex($className) { |
| 1007 |
// Check if class name has numeric index at the end |
| 1008 |
if (preg_match('/^(.+)\.(\d+)$/', $className, $matches)) { |
| 1009 |
return (int)$matches[2]; // Return index as integer |
| 1010 |
} |
| 1011 |
return null; // Return null if no index found |
| 1012 |
} |
| 1013 |
|
| 1014 |
/** |
| 1015 |
* Replace the inner content of tags with given class names in the HTML using DOMDocument. |
| 1016 |
* Supports indexed class names (e.g., "eb-feature-list-title.0", "eb-feature-list-title.1"). |
| 1017 |
* |
| 1018 |
* Note: While CSS selectors would be more readable, PHP's DOMDocument doesn't natively support |
| 1019 |
* CSS selectors. We use XPath which is the standard way to query DOM elements in PHP. |
| 1020 |
* For CSS selector support, you would need a third-party library like symfony/css-selector |
| 1021 |
* or QueryPath, but we keep this implementation dependency-free. |
| 1022 |
* |
| 1023 |
* @param string $html The HTML string. |
| 1024 |
* @param array $contents Array of ['attribute' => className, 'content' => newContent] |
| 1025 |
* @return string The updated HTML. |
| 1026 |
*/ |
| 1027 |
public static function replaceContentByClassNameDom($html, $contents) { |
| 1028 |
$dom = new \DOMDocument(); |
| 1029 |
// Suppress errors due to HTML5 tags or fragments |
| 1030 |
$html = self::escapeInvalidEntities($html); |
| 1031 |
@$dom->loadHTML('<?xml encoding="utf-8" ?>' . $html, LIBXML_HTML_NOIMPLIED | LIBXML_HTML_NODEFDTD); |
| 1032 |
|
| 1033 |
$xpath = new \DOMXPath($dom); |
| 1034 |
foreach ($contents as $item) { |
| 1035 |
$className = $item['attribute']; |
| 1036 |
$newContent = self::escapeInvalidEntities($item['content']); |
| 1037 |
// $newContent = $item['content']; |
| 1038 |
|
| 1039 |
// Extract base class name and index |
| 1040 |
$baseClassName = self::extractBaseClassName($className); |
| 1041 |
$targetIndex = self::extractClassIndex($className); |
| 1042 |
|
| 1043 |
// Find elements by base class name using XPath |
| 1044 |
// XPath equivalent to CSS selector: .baseClassName |
| 1045 |
$nodes = $xpath->query("//*[contains(concat(' ', normalize-space(@class), ' '), ' $baseClassName ')]"); |
| 1046 |
|
| 1047 |
if ($targetIndex !== null) { |
| 1048 |
// If indexed, only replace the element at the specific index |
| 1049 |
if (isset($nodes[$targetIndex])) { |
| 1050 |
$nodes[$targetIndex]->nodeValue = $newContent; |
| 1051 |
} |
| 1052 |
} else { |
| 1053 |
// If not indexed, replace all elements with the class |
| 1054 |
foreach ($nodes as $node) { |
| 1055 |
$node->nodeValue = $newContent; |
| 1056 |
} |
| 1057 |
} |
| 1058 |
} |
| 1059 |
// Remove the XML encoding declaration |
| 1060 |
$result = $dom->saveHTML(); |
| 1061 |
$result = preg_replace('/^<\?xml.*?\?>/', '', $result); |
| 1062 |
return $result; |
| 1063 |
} |
| 1064 |
|
| 1065 |
/** |
| 1066 |
* Replace the inner content of tags with given class names in the HTML using regex. |
| 1067 |
* Supports indexed class names (e.g., "eb-feature-list-title.0", "eb-feature-list-title.1"). |
| 1068 |
* |
| 1069 |
* @param string $html The HTML string. |
| 1070 |
* @param array $contents Array of ['attribute' => className, 'content' => newContent] |
| 1071 |
* @return string The updated HTML. |
| 1072 |
*/ |
| 1073 |
public static function replaceContentByClassNameRegex($html, $contents) { |
| 1074 |
foreach ($contents as $item) { |
| 1075 |
$className = $item['attribute']; |
| 1076 |
$newContent = $item['content']; |
| 1077 |
|
| 1078 |
// Extract base class name and index |
| 1079 |
$baseClassName = self::extractBaseClassName($className); |
| 1080 |
$targetIndex = self::extractClassIndex($className); |
| 1081 |
|
| 1082 |
if ($targetIndex !== null) { |
| 1083 |
// Handle indexed replacement |
| 1084 |
$html = self::replaceContentByClassNameRegexIndexed($html, $baseClassName, $newContent, $targetIndex); |
| 1085 |
} else { |
| 1086 |
// Handle non-indexed replacement (original behavior) |
| 1087 |
$quotedClassName = preg_quote($className, '/'); |
| 1088 |
$pattern = '/(<([a-z0-9]+)[^>]*class="[^"]*\b' . $quotedClassName . '\b[^"]*"[^>]*>)(.*?)(<\/\2>)/is'; |
| 1089 |
$replacement = '$1' . $newContent . '$4'; |
| 1090 |
$html = preg_replace($pattern, $replacement, $html); |
| 1091 |
} |
| 1092 |
} |
| 1093 |
return $html; |
| 1094 |
} |
| 1095 |
|
| 1096 |
/** |
| 1097 |
* Replace content for a specific indexed occurrence of a class name using regex. |
| 1098 |
* |
| 1099 |
* @param string $html The HTML string. |
| 1100 |
* @param string $baseClassName The base class name (without index). |
| 1101 |
* @param string $newContent The new content to replace. |
| 1102 |
* @param int $targetIndex The zero-based index of the element to replace. |
| 1103 |
* @return string The updated HTML. |
| 1104 |
*/ |
| 1105 |
public static function replaceContentByClassNameRegexIndexed($html, $baseClassName, $newContent, $targetIndex) { |
| 1106 |
$quotedClassName = preg_quote($baseClassName, '/'); |
| 1107 |
/* |
| 1108 |
Regex explanation: |
| 1109 |
- (<([a-z0-9]+)[^>]*class="[^"]*\b$baseClassName\b[^"]*"[^>]*>) |
| 1110 |
- (<([a-z0-9]+)[^>]* ... >) : Captures the opening tag with any attributes |
| 1111 |
- ([a-z0-9]+) : Captures the tag name (e.g., p, div, span) |
| 1112 |
- class="[^"]*\b$baseClassName\b[^"]*" : Ensures the class attribute contains the exact base class name (word boundary) |
| 1113 |
- (.*?) : Captures everything inside the tag (non-greedy) |
| 1114 |
- (<\/\2>) : Matches the corresponding closing tag (\2 is the tag name from earlier) |
| 1115 |
Flags: |
| 1116 |
- i : case-insensitive (for tag names) |
| 1117 |
- s : dot matches newlines |
| 1118 |
*/ |
| 1119 |
$pattern = '/(<([a-z0-9]+)[^>]*class="[^"]*\b' . $quotedClassName . '\b[^"]*"[^>]*>)(.*?)(<\/\2>)/is'; |
| 1120 |
|
| 1121 |
$currentIndex = 0; |
| 1122 |
$result = preg_replace_callback($pattern, function($matches) use ($newContent, $targetIndex, &$currentIndex) { |
| 1123 |
if ($currentIndex == $targetIndex) { |
| 1124 |
$currentIndex++; |
| 1125 |
return $matches[1] . $newContent . $matches[4]; |
| 1126 |
} |
| 1127 |
$currentIndex++; |
| 1128 |
return $matches[0]; // Return original match unchanged |
| 1129 |
}, $html); |
| 1130 |
|
| 1131 |
return $result; |
| 1132 |
} |
| 1133 |
|
| 1134 |
/** |
| 1135 |
* Clean block name by removing namespace/plugin prefix |
| 1136 |
* |
| 1137 |
* @param string $block_name The full block name |
| 1138 |
* |
| 1139 |
* @return string Cleaned block name without prefix |
| 1140 |
*/ |
| 1141 |
public static function cleanBlockName( $block_name ) { |
| 1142 |
// Remove namespace/plugin prefix (everything before the last slash) |
| 1143 |
$parts = explode( '/', $block_name ); |
| 1144 |
|
| 1145 |
return end( $parts ); |
| 1146 |
} |
| 1147 |
|
| 1148 |
/** |
| 1149 |
* Escape invalid entities in HTML to prevent DOMDocument warnings. |
| 1150 |
* |
| 1151 |
* @param string $html The HTML string to escape. |
| 1152 |
* @return string The escaped HTML string. |
| 1153 |
*/ |
| 1154 |
public static function escapeInvalidEntities($html) { |
| 1155 |
// Replace & not followed by one of: #, a-z, A-Z, or 0-9, and then a semicolon |
| 1156 |
return preg_replace('/&(?!(#[0-9]+|[a-zA-Z0-9]+);)/', '&', $html); |
| 1157 |
} |
| 1158 |
|
| 1159 |
/** |
| 1160 |
* Remove invalid blocks from array |
| 1161 |
* |
| 1162 |
* @param array $blocks Array of blocks to clean |
| 1163 |
* @return array Cleaned array with only valid blocks |
| 1164 |
*/ |
| 1165 |
public static function cleanInvalidBlocks(array $blocks) { |
| 1166 |
$cleanedBlocks = []; |
| 1167 |
|
| 1168 |
foreach ($blocks as $block) { |
| 1169 |
// Skip if not array |
| 1170 |
if (!is_array($block)) { |
| 1171 |
continue; |
| 1172 |
} |
| 1173 |
|
| 1174 |
// Skip if blockName is null or empty |
| 1175 |
if (empty($block['blockName'])) { |
| 1176 |
continue; |
| 1177 |
} |
| 1178 |
|
| 1179 |
// Skip if missing required properties |
| 1180 |
if (!isset($block['attrs']) || |
| 1181 |
!isset($block['innerBlocks']) || |
| 1182 |
!isset($block['innerHTML']) || |
| 1183 |
!isset($block['innerContent'])) { |
| 1184 |
continue; |
| 1185 |
} |
| 1186 |
|
| 1187 |
// Clean nested blocks recursively |
| 1188 |
if (!empty($block['innerBlocks']) && is_array($block['innerBlocks'])) { |
| 1189 |
$block['innerBlocks'] = self::cleanInvalidBlocks($block['innerBlocks']); |
| 1190 |
} |
| 1191 |
|
| 1192 |
$cleanedBlocks[] = $block; |
| 1193 |
} |
| 1194 |
|
| 1195 |
return $cleanedBlocks; |
| 1196 |
} |
| 1197 |
} |
| 1198 |
|