| @@ -225,14 +225,21 @@ | ||
| 225 | 225 | * |
| 226 | 226 | * `wptexturize()` encodes the characters titles are full of — `&` as |
| 227 | 227 | * `&`, an apostrophe as `’` — and the shell writes titles |
| 228 | 228 | * into text nodes, where the entity renders as itself. Same reasoning |
| 229 | - * as {@see openstation_site_title()}, one layer down. | |
| 229 | + * as {@see openstation_site_title()}, one layer down. Stored names need | |
| 230 | + * it too: a display name, term name or comment author is saved with | |
| 231 | + * `&` as `&`, and so is a title kses filtered on save. | |
| 230 | 232 | * |
| 231 | 233 | * Decode BEFORE the tag strip, never after: `<script>` decodes |
| 232 | 234 | * into a real tag, and stripping second is what removes it. |
| 233 | 235 | * |
| 234 | - * @param string $rendered A title that has been through a display filter. | |
| 236 | + * A `<` opens a tag only before an ASCII letter, `/`, `!` or `?`, the | |
| 237 | + * HTML tokenizer's rule, whether or not a `>` closes it. Any other `<` | |
| 238 | + * is text (`I <3 WordPress`), and `strip_tags()` on its own would drop | |
| 239 | + * it with everything after it. | |
| 240 | + * | |
| 241 | + * @param string $rendered A title or name, rendered or as stored. | |
| 235 | 242 | * @return string Plain text, tag-free. |
| 236 | 243 | */ |
| 237 | 244 | function openstation_plain_text_title( $rendered ) { |
| 238 | 245 | $decoded = html_entity_decode( |
| @@ -240,9 +247,37 @@ | ||
| 240 | 247 | ENT_QUOTES, |
| 241 | 248 | get_bloginfo( 'charset' ) |
| 242 | 249 | ); |
| 243 | 250 | |
| 244 | - return trim( wp_strip_all_tags( $decoded ) ); | |
| 251 | + // A `<` that opens no tag sits out the strip as `<`. `&` is | |
| 252 | + // escaped first and restored last, so an `<` the decode left as | |
| 253 | + // text (`&lt;` in the source) does not come back as a `<` too. | |
| 254 | + $tag_start = '[a-zA-Z\/!?]'; | |
| 255 | + $text = str_replace( '&', '&', $decoded ); | |
| 256 | + $text = openstation_strip_all_tags( $text ); | |
| 257 | + $text = str_replace( '<', '<', $text ); | |
| 258 | + $text = str_replace( '&', '&', $text ); | |
| 259 | + | |
| 260 | + // Removing a tag can leave a kept `<` against the text after it | |
| 261 | + // (`<<b>script>`): a space stops the pair from reading as a tag. | |
| 262 | + return trim( preg_replace( "/<(?={$tag_start})/", '< ', $text ) ); | |
| 263 | +} | |
| 264 | + | |
| 265 | +/** | |
| 266 | + * Strip the tags from stored HTML, keeping a `<` that opens no tag. | |
| 267 | + * | |
| 268 | + * A `<` opens a tag only before an ASCII letter, `/`, `!` or `?`. Any | |
| 269 | + * other one comes back as `<`, the form kses saves it in, so the | |
| 270 | + * result is still HTML text: `wp_trim_words()` and `wp_html_excerpt()`, | |
| 271 | + * which strip tags themselves, pass it on to whatever decodes last. | |
| 272 | + * | |
| 273 | + * @param string $html Stored HTML. | |
| 274 | + * @return string Tag-free text, entities still encoded. | |
| 275 | + */ | |
| 276 | +function openstation_strip_all_tags( $html ) { | |
| 277 | + return wp_strip_all_tags( | |
| 278 | + preg_replace( '/<(?![a-zA-Z\/!?])/', '<', (string) $html ) | |
| 279 | + ); | |
| 245 | 280 | } |
| 246 | 281 | |
| 247 | 282 | /** |
| 248 | 283 | * Build a `WP_Error` for a openstation registration failure. |