| 1 |
<?php |
| 2 |
|
| 3 |
namespace Hostinger\LlmsTxtGenerator; |
| 4 |
|
| 5 |
use WP_Post; |
| 6 |
|
| 7 |
defined( 'ABSPATH' ) || exit; |
| 8 |
|
| 9 |
class LlmsTxtSummaryProvider { |
| 10 |
|
| 11 |
public const TARGET_LENGTH = 160; |
| 12 |
|
| 13 |
public const MINIMUM_WORDS = 4; |
| 14 |
public const MINIMUM_CHARACTERS = 15; |
| 15 |
|
| 16 |
protected const HEADING_PATTERN = '#<h[1-6][^>]*>.*?</h[1-6]>#is'; |
| 17 |
protected const BLOCK_BOUNDARY_PATTERN = '#</?(?:p|h[1-6]|li|dd|dt|div|section|article|aside|header|footer|blockquote|figcaption|td|th)[^>]*>|<br\s*/?>#i'; |
| 18 |
protected const SENTENCE_BOUNDARY_PATTERN = '/(?<=[.!?])\s+|(?<=[.!?]["\')\]])\s+|(?<=[。!?])/u'; |
| 19 |
protected const SENTENCE_END_PATTERN = '/\p{L}.*[.!?。!?]["\')\]]?$/u'; |
| 20 |
protected const DECODED_MARKUP_PATTERN = '#</?[a-z][a-z0-9]*(?:\s+[a-z-]+=(?:"[^"]*"|\'[^\']*\'|[^\s"\'>]+))*\s*/?>#i'; |
| 21 |
protected const WHITESPACE_PATTERN = '/[\s\p{Z}]+/u'; |
| 22 |
|
| 23 |
protected const LEGACY_PUNCTUATION_ENCODING = 'Windows-1252'; |
| 24 |
|
| 25 |
protected const UTF8_SEQUENCE_PATTERN = '/ |
| 26 |
[\x00-\x7F] |
| 27 |
| [\xC2-\xDF][\x80-\xBF] |
| 28 |
| \xE0[\xA0-\xBF][\x80-\xBF] |
| 29 |
| [\xE1-\xEC\xEE\xEF][\x80-\xBF]{2} |
| 30 |
| \xED[\x80-\x9F][\x80-\xBF] |
| 31 |
| \xF0[\x90-\xBF][\x80-\xBF]{2} |
| 32 |
| [\xF1-\xF3][\x80-\xBF]{3} |
| 33 |
| \xF4[\x80-\x8F][\x80-\xBF]{2} |
| 34 |
| (?<invalid>.) |
| 35 |
/xs'; |
| 36 |
|
| 37 |
public function get_summary( WP_Post $post ): string { |
| 38 |
$excerpt = $this->to_plain_text( $post->post_excerpt ); |
| 39 |
|
| 40 |
if ( $excerpt !== '' ) { |
| 41 |
return $excerpt; |
| 42 |
} |
| 43 |
|
| 44 |
return $this->get_content_summary( $post ); |
| 45 |
} |
| 46 |
|
| 47 |
protected function get_content_summary( WP_Post $post ): string { |
| 48 |
$body_html = $this->replace_or_keep( self::HEADING_PATTERN, '', do_blocks( $post->post_content ) ); |
| 49 |
|
| 50 |
foreach ( $this->split_into_text_blocks( $body_html ) as $text_block ) { |
| 51 |
$summary = $this->take_leading_sentences( $text_block ); |
| 52 |
|
| 53 |
if ( $summary !== '' ) { |
| 54 |
return $summary; |
| 55 |
} |
| 56 |
} |
| 57 |
|
| 58 |
return ''; |
| 59 |
} |
| 60 |
|
| 61 |
protected function take_leading_sentences( string $text ): string { |
| 62 |
$sentences = preg_split( self::SENTENCE_BOUNDARY_PATTERN, $text, -1, PREG_SPLIT_OFFSET_CAPTURE ); |
| 63 |
|
| 64 |
if ( ! is_array( $sentences ) ) { |
| 65 |
return ''; |
| 66 |
} |
| 67 |
|
| 68 |
$summary = ''; |
| 69 |
|
| 70 |
foreach ( $sentences as $sentence ) { |
| 71 |
list( $sentence_text, $sentence_offset ) = $sentence; |
| 72 |
|
| 73 |
if ( ! $this->is_complete_sentence( $sentence_text ) ) { |
| 74 |
break; |
| 75 |
} |
| 76 |
|
| 77 |
$candidate = trim( substr( $text, 0, $sentence_offset + strlen( $sentence_text ) ) ); |
| 78 |
|
| 79 |
if ( $this->has_enough_text( $summary ) && $this->character_count( $candidate ) > self::TARGET_LENGTH ) { |
| 80 |
break; |
| 81 |
} |
| 82 |
|
| 83 |
$summary = $candidate; |
| 84 |
} |
| 85 |
|
| 86 |
return $this->has_enough_text( $summary ) ? $summary : ''; |
| 87 |
} |
| 88 |
|
| 89 |
protected function is_complete_sentence( string $sentence ): bool { |
| 90 |
return (bool) preg_match( self::SENTENCE_END_PATTERN, $sentence ); |
| 91 |
} |
| 92 |
|
| 93 |
protected function has_enough_text( string $summary ): bool { |
| 94 |
if ( $this->character_count( $summary ) >= self::MINIMUM_CHARACTERS ) { |
| 95 |
return true; |
| 96 |
} |
| 97 |
|
| 98 |
$words = preg_split( self::WHITESPACE_PATTERN, $summary, -1, PREG_SPLIT_NO_EMPTY ); |
| 99 |
|
| 100 |
if ( ! is_array( $words ) ) { |
| 101 |
return false; |
| 102 |
} |
| 103 |
|
| 104 |
return count( $words ) >= self::MINIMUM_WORDS; |
| 105 |
} |
| 106 |
|
| 107 |
/** |
| 108 |
* Core's mb_strlen() polyfill counts bytes unless it is given an encoding. |
| 109 |
*/ |
| 110 |
protected function character_count( string $text ): int { |
| 111 |
return mb_strlen( $text, 'UTF-8' ); |
| 112 |
} |
| 113 |
|
| 114 |
protected function to_plain_text( string $html ): string { |
| 115 |
return implode( ' ', $this->split_into_text_blocks( $html ) ); |
| 116 |
} |
| 117 |
|
| 118 |
protected function split_into_text_blocks( string $html ): array { |
| 119 |
$chunks = preg_split( self::BLOCK_BOUNDARY_PATTERN, $html ); |
| 120 |
|
| 121 |
if ( ! is_array( $chunks ) ) { |
| 122 |
return array(); |
| 123 |
} |
| 124 |
|
| 125 |
$text_blocks = array(); |
| 126 |
|
| 127 |
foreach ( $chunks as $chunk ) { |
| 128 |
$text = $this->strip_markup( $chunk ); |
| 129 |
|
| 130 |
if ( $text !== '' ) { |
| 131 |
$text_blocks[] = $text; |
| 132 |
} |
| 133 |
} |
| 134 |
|
| 135 |
return $text_blocks; |
| 136 |
} |
| 137 |
|
| 138 |
protected function strip_markup( string $html ): string { |
| 139 |
$text = $this->to_valid_utf8( $html ); |
| 140 |
$decoded = html_entity_decode( wp_strip_all_tags( strip_shortcodes( $text ) ), ENT_QUOTES | ENT_HTML5, 'UTF-8' ); |
| 141 |
$plain = $this->replace_or_keep( self::DECODED_MARKUP_PATTERN, '', $decoded ); |
| 142 |
|
| 143 |
return trim( $this->replace_or_keep( self::WHITESPACE_PATTERN, ' ', $plain ) ); |
| 144 |
} |
| 145 |
|
| 146 |
protected function to_valid_utf8( string $text ): string { |
| 147 |
if ( $this->is_valid_utf8( $text ) ) { |
| 148 |
return $text; |
| 149 |
} |
| 150 |
|
| 151 |
if ( function_exists( 'mb_convert_encoding' ) ) { |
| 152 |
$repaired = $this->repair_invalid_bytes( $text ); |
| 153 |
|
| 154 |
if ( $repaired !== '' && $this->is_valid_utf8( $repaired ) ) { |
| 155 |
return $repaired; |
| 156 |
} |
| 157 |
} |
| 158 |
|
| 159 |
return wp_check_invalid_utf8( $text, true ); |
| 160 |
} |
| 161 |
|
| 162 |
/** |
| 163 |
* Only the invalid bytes are converted; converting the whole string would re-encode |
| 164 |
* punctuation that is already correct. |
| 165 |
*/ |
| 166 |
protected function repair_invalid_bytes( string $text ): string { |
| 167 |
$repaired = preg_replace_callback( |
| 168 |
self::UTF8_SEQUENCE_PATTERN, |
| 169 |
static function ( array $matches ): string { |
| 170 |
if ( ! isset( $matches['invalid'] ) ) { |
| 171 |
return $matches[0]; |
| 172 |
} |
| 173 |
|
| 174 |
return (string) mb_convert_encoding( $matches['invalid'], 'UTF-8', self::LEGACY_PUNCTUATION_ENCODING ); |
| 175 |
}, |
| 176 |
$text |
| 177 |
); |
| 178 |
|
| 179 |
return $repaired ?? $text; |
| 180 |
} |
| 181 |
|
| 182 |
/** |
| 183 |
* wp_check_invalid_utf8() answers on the site's blog_charset rather than on the bytes, |
| 184 |
* and wp_is_valid_utf8() only exists from WordPress 6.9 while this plugin supports 5.5. |
| 185 |
*/ |
| 186 |
protected function is_valid_utf8( string $text ): bool { |
| 187 |
return preg_match( '//u', $text ) === 1; |
| 188 |
} |
| 189 |
|
| 190 |
protected function replace_or_keep( string $pattern, string $replacement, string $subject ): string { |
| 191 |
return preg_replace( $pattern, $replacement, $subject ) ?? $subject; |
| 192 |
} |
| 193 |
} |
| 194 |
|