| @@ -140,8 +140,9 @@ | ||
| 140 | 140 | 'grade' => $result['grade'], |
| 141 | 141 | ]]; |
| 142 | 142 | $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords); |
| 143 | 143 | } |
| 144 | + $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords); | |
| 144 | 145 | |
| 145 | 146 | return $result; |
| 146 | 147 | } |
| 147 | 148 | |
| @@ -251,8 +252,9 @@ | ||
| 251 | 252 | $best['target_keyword'] = $best_keyword; |
| 252 | 253 | $best['target_keywords'] = $keywords; |
| 253 | 254 | $best['keyword_results'] = $per_keyword; |
| 254 | 255 | $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords); |
| 256 | + $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords); | |
| 255 | 257 | |
| 256 | 258 | return $best; |
| 257 | 259 | } |
| 258 | 260 | |
| @@ -267,15 +269,15 @@ | ||
| 267 | 269 | * @param string[] $keywords Target keywords. |
| 268 | 270 | * @return array<string,array{passed:bool,matched_keywords:string[]}> |
| 269 | 271 | */ |
| 270 | 272 | private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array { |
| 271 | - $title = strtolower((string) ($metadata['title'] ?? $content_data['title'] ?? '')); | |
| 272 | - $description = strtolower((string) ($metadata['description'] ?? '')); | |
| 273 | - $content = strtolower(wp_strip_all_tags((string) ($content_data['content'] ?? ''))); | |
| 273 | + $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')); | |
| 274 | + $description = self::lower((string) ($metadata['description'] ?? '')); | |
| 275 | + $content = self::lower(self::plain_text((string) ($content_data['content'] ?? ''))); | |
| 274 | 276 | |
| 275 | 277 | $alts = ''; |
| 276 | 278 | foreach ((array) ($content_data['images'] ?? []) as $image) { |
| 277 | - $alts .= ' ' . strtolower((string) ($image['alt'] ?? '')); | |
| 279 | + $alts .= ' ' . self::lower((string) ($image['alt'] ?? '')); | |
| 278 | 280 | } |
| 279 | 281 | |
| 280 | 282 | // Build a searchable slug haystack from the post's OWN slug — never the |
| 281 | 283 | // full URL path. The path carries ancestors, category bases and date |
| @@ -284,17 +286,9 @@ | ||
| 284 | 286 | // way: an unpublished post has no pretty permalink (get_permalink() |
| 285 | 287 | // returns ?p=123), so the path held no slug at all and every draft |
| 286 | 288 | // scored "no match" until it was published. Hyphens/underscores become |
| 287 | 289 | // spaces so multi-word keywords can match. |
| 288 | - $slug_source = (string) ($content_data['slug'] ?? ''); | |
| 289 | - if ($slug_source === '') { | |
| 290 | - // Draft with no slug assigned yet: score what WordPress would | |
| 291 | - // generate from the title, which is what the editor shows as the | |
| 292 | - // proposed URL — so the check reads the same before and after | |
| 293 | - // publishing instead of flipping. | |
| 294 | - $slug_source = sanitize_title((string) ($content_data['title'] ?? '')); | |
| 295 | - } | |
| 296 | - $slug = strtolower(str_replace(['-', '_'], ' ', $slug_source)); | |
| 290 | + $slug = self::lower(self::slug_haystack($content_data)); | |
| 297 | 291 | |
| 298 | 292 | $haystacks = [ |
| 299 | 293 | 'title' => trim($title), |
| 300 | 294 | 'meta_description' => trim($description), |
| @@ -306,9 +300,9 @@ | ||
| 306 | 300 | $checks = []; |
| 307 | 301 | foreach ($haystacks as $location => $haystack) { |
| 308 | 302 | $matched = []; |
| 309 | 303 | foreach ($keywords as $keyword) { |
| 310 | - if ($this->keyword_matches($haystack, strtolower(trim($keyword)))) { | |
| 304 | + if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) { | |
| 311 | 305 | $matched[] = $keyword; |
| 312 | 306 | } |
| 313 | 307 | } |
| 314 | 308 | $checks[$location] = [ |
| @@ -317,8 +311,210 @@ | ||
| 317 | 311 | ]; |
| 318 | 312 | } |
| 319 | 313 | |
| 320 | 314 | return $checks; |
| 315 | + } | |
| 316 | + | |
| 317 | + /** | |
| 318 | + * The keyword placements the editor draws one gauge segment for, in the | |
| 319 | + * order a reader meets them (#729). | |
| 320 | + * | |
| 321 | + * @since 2.11.0 | |
| 322 | + * @var string[] | |
| 323 | + */ | |
| 324 | + public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt']; | |
| 325 | + | |
| 326 | + /** | |
| 327 | + * Characters of plain text read as the opening when the content has no | |
| 328 | + * paragraph tag. | |
| 329 | + * | |
| 330 | + * @since 2.11.0 | |
| 331 | + */ | |
| 332 | + private const OPENING_CHARS = 300; | |
| 333 | + | |
| 334 | + /** | |
| 335 | + * Where each focus keyword is placed, keyword by keyword (#729). | |
| 336 | + * | |
| 337 | + * analyze_keyword_checks() answers "does ANY keyword appear here" for | |
| 338 | + * five places; this answers "where does THIS keyword appear" for seven, | |
| 339 | + * so the editor can show each keyword's own gauge. Same matcher, so a | |
| 340 | + * keyword counts as a word (not a fragment) and a keyword in a script | |
| 341 | + * written without spaces (Thai, Chinese, Japanese) still matches. | |
| 342 | + * | |
| 343 | + * `where` names the heading or alt text that matched, so the editor can | |
| 344 | + * say which one. | |
| 345 | + * | |
| 346 | + * @since 2.11.0 | |
| 347 | + * | |
| 348 | + * @param array $content_data Content analysis data. | |
| 349 | + * @param array $metadata Post metadata (title, description). | |
| 350 | + * @param string[] $keywords Focus keywords. | |
| 351 | + * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}> | |
| 352 | + */ | |
| 353 | + public function keyword_placements(array $content_data, array $metadata, array $keywords): array { | |
| 354 | + $html = (string) ($content_data['content'] ?? ''); | |
| 355 | + $plain = self::lower(self::plain_text($html)); | |
| 356 | + | |
| 357 | + $headings = []; | |
| 358 | + $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html); | |
| 359 | + foreach ($source as $heading) { | |
| 360 | + $text = trim((string) ($heading['text'] ?? '')); | |
| 361 | + if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) { | |
| 362 | + $headings[] = $text; | |
| 363 | + } | |
| 364 | + } | |
| 365 | + | |
| 366 | + $alts = []; | |
| 367 | + foreach ((array) ($content_data['images'] ?? []) as $image) { | |
| 368 | + $alt = trim((string) ($image['alt'] ?? '')); | |
| 369 | + if ('' !== $alt) { | |
| 370 | + $alts[] = $alt; | |
| 371 | + } | |
| 372 | + } | |
| 373 | + | |
| 374 | + $single = [ | |
| 375 | + 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')), | |
| 376 | + 'meta_description' => self::lower((string) ($metadata['description'] ?? '')), | |
| 377 | + 'slug' => self::lower(self::slug_haystack($content_data)), | |
| 378 | + 'first_paragraph' => self::lower(self::opening($html)), | |
| 379 | + 'content' => $plain, | |
| 380 | + ]; | |
| 381 | + | |
| 382 | + $out = []; | |
| 383 | + foreach ($keywords as $keyword) { | |
| 384 | + $needle = self::lower(trim((string) $keyword)); | |
| 385 | + $placements = []; | |
| 386 | + | |
| 387 | + foreach (self::PLACEMENTS as $placement) { | |
| 388 | + if ('subheading' === $placement || 'image_alt' === $placement) { | |
| 389 | + $where = ''; | |
| 390 | + foreach ('subheading' === $placement ? $headings : $alts as $text) { | |
| 391 | + if ($this->keyword_matches(self::lower($text), $needle)) { | |
| 392 | + $where = $text; | |
| 393 | + break; | |
| 394 | + } | |
| 395 | + } | |
| 396 | + $placements[$placement] = ['passed' => '' !== $where, 'where' => $where]; | |
| 397 | + continue; | |
| 398 | + } | |
| 399 | + | |
| 400 | + $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => '']; | |
| 401 | + } | |
| 402 | + | |
| 403 | + $out[] = [ | |
| 404 | + 'keyword' => (string) $keyword, | |
| 405 | + 'passed' => count(array_filter(array_column($placements, 'passed'))), | |
| 406 | + 'total' => count(self::PLACEMENTS), | |
| 407 | + 'placements' => $placements, | |
| 408 | + ]; | |
| 409 | + } | |
| 410 | + | |
| 411 | + return $out; | |
| 412 | + } | |
| 413 | + | |
| 414 | + /** | |
| 415 | + * The post's own slug as searchable text: hyphens and underscores become | |
| 416 | + * spaces so a multi-word keyword can match, and a slug WordPress | |
| 417 | + * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never | |
| 418 | + * match a keyword typed in that script. | |
| 419 | + * | |
| 420 | + * Never the full URL path: the path carries ancestors, category bases and | |
| 421 | + * date segments, so a child of /clinical-trials/ reported "keyword in slug" | |
| 422 | + * for a page actually slugged `contact-us`. An unpublished post has no | |
| 423 | + * pretty permalink either, so the path held no slug at all. | |
| 424 | + * | |
| 425 | + * @param array $content_data Content analysis data. | |
| 426 | + * @return string | |
| 427 | + */ | |
| 428 | + private static function slug_haystack(array $content_data): string { | |
| 429 | + $slug = (string) ($content_data['slug'] ?? ''); | |
| 430 | + if ('' === $slug) { | |
| 431 | + // Draft with no slug assigned yet: score what WordPress would | |
| 432 | + // generate from the title, which is what the editor shows as the | |
| 433 | + // proposed URL — so the check reads the same before and after | |
| 434 | + // publishing instead of flipping. | |
| 435 | + $slug = sanitize_title((string) ($content_data['title'] ?? '')); | |
| 436 | + } | |
| 437 | + | |
| 438 | + return trim(str_replace(['-', '_'], ' ', rawurldecode($slug))); | |
| 439 | + } | |
| 440 | + | |
| 441 | + /** | |
| 442 | + * The opening of the content: its first paragraph with text, or the first | |
| 443 | + * few hundred characters when it has none. Counted in characters, not | |
| 444 | + * words, so a language written without spaces is not read as one word. | |
| 445 | + * | |
| 446 | + * @param string $html Content HTML. | |
| 447 | + * @return string Plain text. | |
| 448 | + */ | |
| 449 | + private static function opening(string $html): string { | |
| 450 | + if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) { | |
| 451 | + foreach ($matches[1] as $paragraph) { | |
| 452 | + $text = self::collapse_whitespace(wp_strip_all_tags($paragraph)); | |
| 453 | + if ('' !== $text) { | |
| 454 | + return $text; | |
| 455 | + } | |
| 456 | + } | |
| 457 | + } | |
| 458 | + | |
| 459 | + $text = self::plain_text($html); | |
| 460 | + | |
| 461 | + return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS); | |
| 462 | + } | |
| 463 | + | |
| 464 | + /** | |
| 465 | + * Content as plain text, with a space where each tag was. wp_strip_all_tags() | |
| 466 | + * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became | |
| 467 | + * "coffee grinderA good" — so a keyword at the end of a heading or a | |
| 468 | + * paragraph was no longer a word and did not match. | |
| 469 | + * | |
| 470 | + * @param string $html Content HTML. | |
| 471 | + * @return string | |
| 472 | + */ | |
| 473 | + private static function plain_text(string $html): string { | |
| 474 | + $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html); | |
| 475 | + | |
| 476 | + return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced)); | |
| 477 | + } | |
| 478 | + | |
| 479 | + /** | |
| 480 | + * Runs of whitespace down to one space. | |
| 481 | + * | |
| 482 | + * The `/u` pass is the one that understands a multibyte space, but | |
| 483 | + * preg_replace() answers null on bytes that are not valid UTF-8 rather than | |
| 484 | + * throwing — and casting that null to a string blanked the haystack, so a | |
| 485 | + * post carrying one mojibake byte (a Latin-1 paste, an old import) reported | |
| 486 | + * every keyword as missing from its body, its opening, and every | |
| 487 | + * subheading. The gauge said 0/7 and told the author to add a keyword that | |
| 488 | + * was already there. | |
| 489 | + * | |
| 490 | + * Falls back to the byte-wise collapse, which is what this did before the | |
| 491 | + * multibyte work added the modifier. Same reasoning keyword_matches() | |
| 492 | + * already records for its own PCRE failure: a pattern PCRE refuses must not | |
| 493 | + * be reported as a confident "no match". | |
| 494 | + * | |
| 495 | + * @param string $text Text to collapse. | |
| 496 | + * @return string | |
| 497 | + */ | |
| 498 | + private static function collapse_whitespace(string $text): string { | |
| 499 | + $collapsed = preg_replace('/\s+/u', ' ', $text); | |
| 500 | + | |
| 501 | + if (null === $collapsed) { | |
| 502 | + $collapsed = preg_replace('/\s+/', ' ', $text); | |
| 503 | + } | |
| 504 | + | |
| 505 | + return trim(null === $collapsed ? $text : (string) $collapsed); | |
| 506 | + } | |
| 507 | + | |
| 508 | + /** | |
| 509 | + * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never | |
| 510 | + * matched "кофе". | |
| 511 | + * | |
| 512 | + * @param string $text Text. | |
| 513 | + * @return string | |
| 514 | + */ | |
| 515 | + private static function lower(string $text): string { | |
| 516 | + return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text); | |
| 321 | 517 | } |
| 322 | 518 | |
| 323 | 519 | /** |
| 324 | 520 | * Scripts written without spaces between words. |