PluginProbe ʕ •ᴥ•ʔ
Jetpack – WP Security, Backup, Speed, & Growth / 12.1.3
Jetpack – WP Security, Backup, Speed, & Growth v12.1.3
12.0.3 12.1.3 12.2.3 12.3.2 12.4.2 12.5.2 12.6.4 12.7.3 12.8.3 12.9.5 13.0.2 13.1.5 13.2.4 13.3.3 13.4.5 13.5.2 13.6.2 13.7.2 13.8.3 13.9.2 14.0.1 14.1.1 14.2.2 14.3.1 14.4.2 14.5.1 14.6.1 14.7.1 14.8.1 14.9.2 15.0.3 15.1.2 15.2.1 15.3.2 15.4.1 15.5.1 15.6.1 15.7.2 15.8.1 15.9.2 16.0.2 16.1.3 16.2-a.5 16.2-a.3 16.1.2 16.2-a.1 16.1.1 16.1 16.1-beta 16.1-beta.2 16.1-beta.3 16.1-a.5 16.1-a.3 16.0.1 16.1-a.1 16.0 16.0-beta 16.0-a.7 16.0-a.5 15.9.1 16.0-a.3 16.0-a.1 15.9 15.9-beta 15.9-a.7 15.9-a.5 15.9-a.3 15.9-a.1 15.8 15.8-beta 15.8-a.7 15.8-a.5 5.2.5 5.3.4 5.4.4 5.5.5 5.6.5 5.7.5 5.8.4 5.9.4 6.0.4 6.1 6.1.1 6.1.2 6.1.3 6.1.4 6.1.5 6.2 6.2.1 6.2.2 6.2.3 6.2.4 6.2.5 6.3 6.3.1 6.3.2 6.3.3 6.3.4 6.3.5 6.3.6 6.3.7 6.4 6.4.1 6.4.2 6.4.3 6.4.4 6.4.5 6.4.6 6.5 6.5.1 6.5.2 6.5.3 6.5.4 6.6 6.6.1 6.6.2 6.6.3 6.6.4 6.6.5 6.7 6.7.1 6.7.2 6.7.3 6.7.4 6.8 6.8.1 6.8.2 6.8.3 6.8.4 6.8.5 6.9 6.9.1 6.9.2 6.9.3 6.9.4 7.0 7.0.1 7.0.2 7.0.3 7.0.4 7.0.5 7.1 7.1.1 7.1.2 7.1.3 7.1.4 7.1.5 7.2 7.2.1 7.2.1.1 7.2.2 7.2.3 7.2.4 7.2.5 7.3 7.3.0.1 7.3.1 7.3.1.1 7.3.2 7.3.3 7.3.4 7.3.5 7.4 7.4.1 7.4.2 7.4.3 7.4.4 7.4.5 7.5 7.5.0.1 7.5.1 7.5.2 7.5.3 7.5.4 7.5.5 7.5.6 7.5.7 7.6 7.6.1 7.6.2 7.6.3 7.6.4 7.7 7.7.1 7.7.2 7.7.3 7.7.4 7.7.5 7.7.6 7.8 7.8.1 7.8.2 7.8.3 7.8.4 7.9 7.9.1 7.9.2 7.9.3 7.9.4 8.0 8.0.1 8.0.2 8.0.3 8.1 8.1.1 8.1.2 8.1.3 8.1.4 8.2 8.2.0.1 8.2.1 8.2.2 8.2.3 8.2.4 8.2.5 8.2.6 8.3 8.3.1 8.3.2 8.3.3 8.4 8.4.1 8.4.2 8.4.3 8.4.4 8.4.5 8.5 8.5.1 8.5.2 8.5.3 8.6 8.6.1 8.6.2 8.6.3 8.6.4 8.7 8.7.0.1 8.7.1 8.7.2 8.7.3 8.7.4 8.8 8.8.1 8.8.2 8.8.3 8.8.4 8.8.5 8.9 8.9.1 8.9.2 8.9.3 8.9.4 9.0 9.0.1 9.0.2 9.0.3 9.0.4 9.0.5 9.1 9.1.1 9.1.2 9.1.3 9.2 9.2.1 9.2.2 9.2.3 9.2.4 9.3 9.3.1 9.3.2 9.3.3 9.3.4 9.3.5 9.4 9.4.1 9.4.2 9.4.3 9.4.4 9.5 9.5.1 9.5.2 9.5.3 9.5.4 9.5.5 9.6 9.6.1 9.6.2 9.6.3 9.6.4 9.7 9.7.1 9.7.2 15.7-beta.2 9.7.3 15.7.1 9.8 15.8-a.1 9.8.1 15.8-a.3 9.8.2 2.0.9 9.8.3 2.1.7 9.9 2.2.10 9.9.1 2.3.10 9.9.2 2.4.7 9.9.3 2.5.5 2.6.6 2.7.5 2.8.5 2.9.6 3.0.6 3.1.5 3.2.5 3.3.6 3.4.6 3.5.6 3.6.4 3.7.5 3.8.5 3.9.10 4.0.7 4.1.4 4.2.5 4.3.5 4.4.5 4.5.3 4.6.3 4.7.4 4.8.5 4.9.3 5.0.3 5.1.4 trunk 10.0 10.0.1 10.0.2 10.1 10.1.1 10.1.2 10.2 10.2.1 10.2.2 10.2.3 10.3 10.3.1 10.3.2 10.4 10.4.1 10.4.2 10.5 10.5.1 10.5.2 10.5.3 10.6 10.6.1 10.6.2 10.7 10.7.1 10.7.2 10.8 10.8.1 10.8.2 10.9 10.9.1 10.9.2 10.9.3 11.0 11.0.1 11.0.2 11.1 11.1.1 11.1.2 11.1.3 11.1.4 11.2 11.2.1 11.2.2 11.3 11.3.1 11.3.2 11.3.3 11.3.4 11.4 11.4.1 11.4.2 11.5 11.5.1 11.5.2 11.5.3 11.6 11.6.1 11.6.2 11.7 11.7.1 11.7.2 11.7.3 11.8 11.8.3 11.8.4 11.8.5 11.8.6 11.9 11.9.1 11.9.2 11.9.3 12.0 12.0.1 12.0.2 12.1 12.1.1 12.1.2 12.2 12.2.1 12.2.2 12.3 12.3.1 12.4 12.4.1 12.5 12.5.1 12.6 12.6.1 12.6.2 12.6.3 12.7 12.7.1 12.7.2 12.8 12.8.1 12.8.2 12.9 12.9.1 12.9.2 12.9.3 12.9.4 13.0 13.0.1 13.1 13.1.1 13.1.2 13.1.3 13.1.4 13.2 13.2.1 13.2.2 13.2.3 13.3 13.3.1 13.3.2 13.4 13.4.1 13.4.2 13.4.3 13.4.4 13.5 13.5.1 13.6 13.6.1 13.7 13.7.1 13.8 13.8.1 13.8.2 13.9 13.9.1 14.0 14.1 14.2 14.2.1 14.3 14.4 14.4.1 14.5 14.6 14.7 14.8 14.9 14.9.1 15.0 15.0.1 15.0.2 15.1 15.1.1 15.2 15.3 15.3.1 15.4 15.5 15.6 15.7 15.7-a.1 15.7-a.3 15.7-a.5 15.7-a.7 15.7-beta
jetpack / _inc / lib / class-jetpack-tweetstorm-helper.php
jetpack / _inc / lib Last commit date
admin-pages 3 years ago core-api 3 years ago debugger 3 years ago markdown 4 years ago class-jetpack-ai-helper.php 3 years ago class-jetpack-currencies.php 5 years ago class-jetpack-google-drive-helper.php 3 years ago class-jetpack-instagram-gallery-helper.php 3 years ago class-jetpack-mapbox-helper.php 3 years ago class-jetpack-podcast-feed-locator.php 5 years ago class-jetpack-podcast-helper.php 3 years ago class-jetpack-recommendations.php 4 years ago class-jetpack-tweetstorm-helper.php 3 years ago class-jetpack-wizard.php 5 years ago class.color.php 4 years ago class.core-rest-api-endpoints.php 3 years ago class.jetpack-automatic-install-skin.php 4 years ago class.jetpack-iframe-embed.php 4 years ago class.jetpack-keyring-service-helper.php 4 years ago class.jetpack-password-checker.php 3 years ago class.jetpack-photon-image-sizes.php 3 years ago class.jetpack-photon-image.php 4 years ago class.jetpack-search-performance-logger.php 4 years ago class.media-extractor.php 3 years ago class.media-summary.php 3 years ago class.media.php 3 years ago components.php 3 years ago debugger.php 4 years ago functions.wp-notify.php 4 years ago icalendar-reader.php 3 years ago markdown.php 3 years ago plans.php 4 years ago plugins.php 4 years ago tonesque.php 3 years ago widgets.php 4 years ago
class-jetpack-tweetstorm-helper.php
1754 lines
1 <?php
2 /**
3 * Tweetstorm block and API helper.
4 *
5 * @package automattic/jetpack
6 * @since 8.7.0
7 */
8
9 use Automattic\Jetpack\Connection\Client;
10 use Automattic\Jetpack\Connection\Manager;
11 use Automattic\Jetpack\Status;
12 use Twitter\Text\Regex as Twitter_Regex;
13 use Twitter\Text\Validator as Twitter_Validator;
14
15 /**
16 * Class Jetpack_Tweetstorm_Helper
17 *
18 * @since 8.7.0
19 */
20 class Jetpack_Tweetstorm_Helper {
21 /**
22 * Blocks that can be converted to tweets.
23 *
24 * @var array {
25 * The key for each element must match the registered block name.
26 *
27 * @type string $type Required. The type of content this block produces. Can be one of 'break', 'embed', 'image',
28 * 'multiline', 'text', or 'video'.
29 * @type string $content_location Optional. Where the block content can be found. Can be 'html', if we need to parse
30 * it out of the block HTML text, 'html-attributes', if the we need to parse it out of HTML attributes
31 * in the block HTML, or 'block-attributes', if the content can be found in the block attributes.
32 * Note that these attributes need to be available when the serialised block is
33 * parsed using `parse_blocks()`. If it isn't set, it's assumed the block doesn't add
34 * any content to the Twitter thread.
35 * @type array $content Optional. Defines what parts of the block content need to be extracted. Behaviour can vary based on
36 * `$content_location`, and `$type`:
37 *
38 * - When `$content_location` is 'html', a value of `array()` or `array( 'content' )` have the same meaning:
39 * The entire block HTML should be used. In both cases, 'content' will be the corresponding tag in `$template`.
40 * - When `$content_location` is 'html', it should be formatted as `array( 'container' => 'tag' )`,
41 * where 'container' is the name of the corresponding RichText container in the block editor, and is also the name
42 * of the corresponding tag in the $template string. 'tag' is the HTML tag within the block that corresponds to this
43 * container. When `$type` is 'multiline', there must only be one element in the array, and tag should be set to the HTML
44 * tag that corresponds to each line, though the 'container' should still be the RichText container name. (Eg, in the core/list block, the tag is 'li'.)
45 * - When `$content_location` is 'html-attributes', the array should be formatted as `array( 'name' => array( 'tag', 'attribute') )`,
46 * where 'name' is the name of a particular value that different block types require, 'tag' is the name of the HTML tag where 'attribute'
47 * can be found, containing the value to use for 'name'. When `$type` is 'image', 'url' and 'alt' must be defined. When `$type` is 'video',
48 * 'url' must be defined.
49 * - When `$content_location` is 'block-attributes', it must be an array of block attribute names. When `$type` is 'embed', there
50 * only be one element, corresponding to the URL for the embed.
51 * @type string $template Required for 'text' and 'multiline' types, ignored for all other types. Describes how the block content will be formatted when tweeted.
52 * Tags should match the keys of `$content`, except for the special "{{content}}", which matches the entire HTML content of the block.
53 * For 'multiline' types, the template will be repeated for every line in the block.
54 * @type boolean $force_new Required. Whether or not a new tweet should be started when this block is encountered.
55 * @type boolean $force_finished Required. Whether or not a new tweet should be started after this block is finished.
56 * }
57 */
58 private static $supported_blocks = array(
59 'core/embed' => array(
60 'type' => 'embed',
61 'content_location' => 'block-attributes',
62 'content' => array( 'url' ),
63 'force_new' => false,
64 'force_finished' => true,
65 ),
66 'core/gallery' => array(
67 'type' => 'image',
68 'content_location' => 'html-attributes',
69 'content' => array(
70 'url' => array( 'img', 'src' ),
71 'alt' => array( 'img', 'alt' ),
72 ),
73 'force_new' => false,
74 'force_finished' => true,
75 ),
76 'core/heading' => array(
77 'type' => 'text',
78 'content_location' => 'html',
79 'content' => array(),
80 'template' => '{{content}}',
81 'force_new' => true,
82 'force_finished' => false,
83 ),
84 'core/image' => array(
85 'type' => 'image',
86 'content_location' => 'html-attributes',
87 'content' => array(
88 'url' => array( 'img', 'src' ),
89 'alt' => array( 'img', 'alt' ),
90 ),
91 'force_new' => false,
92 'force_finished' => true,
93 ),
94 'core/list' => array(
95 'type' => 'multiline',
96 'content_location' => 'html',
97 // It looks a little weird to use the 'values' key for a single line,
98 // but 'values' is the name of the RichText content area.
99 'content' => array(
100 'values' => 'li',
101 ),
102 'template' => '- {{values}}',
103 'force_new' => false,
104 'force_finished' => false,
105 ),
106 'core/paragraph' => array(
107 'type' => 'text',
108 'content_location' => 'html',
109 'content' => array(),
110 'template' => '{{content}}',
111 'force_new' => false,
112 'force_finished' => false,
113 ),
114 'core/quote' => array(
115 'type' => 'text',
116 'content_location' => 'html',
117 // The quote content will always be inside <p> tags.
118 'content' => array(
119 'value' => 'p',
120 'citation' => 'cite',
121 ),
122 'template' => '“{{value}}” – {{citation}}',
123 'force_new' => false,
124 'force_finished' => false,
125 ),
126 'core/separator' => array(
127 'type' => 'break',
128 'force_new' => false,
129 'force_finished' => true,
130 ),
131 'core/spacer' => array(
132 'type' => 'break',
133 'force_new' => false,
134 'force_finished' => true,
135 ),
136 'core/verse' => array(
137 'type' => 'text',
138 'content_location' => 'html',
139 'content' => array(),
140 'template' => '{{content}}',
141 'force_new' => false,
142 'force_finished' => false,
143 ),
144 'core/video' => array(
145 'type' => 'video',
146 'content_location' => 'html-attributes',
147 'content' => array(
148 'url' => array( 'video', 'src' ),
149 ),
150 'force_new' => false,
151 'force_finished' => true,
152 ),
153 'jetpack/gif' => array(
154 'type' => 'embed',
155 'content_location' => 'block-attributes',
156 'content' => array( 'giphyUrl' ),
157 'force_new' => false,
158 'force_finished' => true,
159 ),
160 );
161
162 /**
163 * A cache of _wp_emoji_list( 'entities' ), after being run through html_entity_decode().
164 *
165 * Initialised in ::is_valid_tweet().
166 *
167 * @var array
168 */
169 private static $emoji_list = array();
170
171 /**
172 * Special line separator character, for multiline text.
173 *
174 * @var string
175 */
176 private static $line_separator = "\xE2\x80\xA8";
177
178 /**
179 * Special inline placeholder character, for inline tags that change content length in the RichText..
180 *
181 * @var string
182 */
183 private static $inline_placeholder = "\xE2\x81\xA3";
184
185 /**
186 * URLs always take up a fixed length from the text limit.
187 *
188 * @var int
189 */
190 private static $characters_per_url = 24;
191
192 /**
193 * Every media attachment takes up some space from the text limit.
194 *
195 * @var int
196 */
197 private static $characters_per_media = 24;
198
199 /**
200 * An array to store all the tweets in.
201 *
202 * @var array
203 */
204 private static $tweets = array();
205
206 /**
207 * While we're caching everything, we want to keep track of the URLs we're adding.
208 *
209 * @var array
210 */
211 private static $urls = array();
212
213 /**
214 * Checks if a given request is allowed to gather tweets.
215 *
216 * @param WP_REST_Request $request Full details about the request.
217 *
218 * @return true|WP_Error True if the request has access to gather tweets from a thread, WP_Error object otherwise.
219 */
220 public static function permissions_check( $request ) { // phpcs:ignore Generic.CodeAnalysis.UnusedFunctionParameter, VariableAnalysis.CodeAnalysis.VariableAnalysis.UnusedVariable
221 $blog_id = get_current_blog_id();
222
223 /*
224 * User hitting the endpoint hosted on their Jetpack site, from their Jetpack site,
225 * or hitting the endpoint hosted on WPCOM, from their WPCOM site.
226 */
227 if ( current_user_can_for_blog( $blog_id, 'edit_posts' ) ) {
228 return true;
229 }
230
231 // Jetpack hitting the endpoint hosted on WPCOM, from a Jetpack site with a blog token.
232 if ( defined( 'IS_WPCOM' ) && IS_WPCOM ) {
233 if ( is_jetpack_site( $blog_id ) ) {
234 if ( ! class_exists( 'WPCOM_REST_API_V2_Endpoint_Jetpack_Auth' ) ) {
235 require_once dirname( __DIR__ ) . '/rest-api-plugins/endpoints/jetpack-auth.php';
236 }
237
238 $jp_auth_endpoint = new WPCOM_REST_API_V2_Endpoint_Jetpack_Auth();
239 if ( true === $jp_auth_endpoint->is_jetpack_authorized_for_site() ) {
240 return true;
241 }
242 }
243 }
244
245 return new WP_Error(
246 'rest_forbidden',
247 __( 'Sorry, you are not allowed to use tweetstorm endpoints on this site.', 'jetpack' ),
248 array( 'status' => rest_authorization_required_code() )
249 );
250 }
251
252 /**
253 * Gather the Tweetstorm.
254 *
255 * @param string $url The tweet URL to gather from.
256 * @return mixed
257 */
258 public static function gather( $url ) {
259 if ( ( new Status() )->is_offline_mode() ) {
260 return new WP_Error(
261 'dev_mode',
262 __( 'Tweet unrolling is not available in offline mode.', 'jetpack' )
263 );
264 }
265
266 $site_id = Manager::get_site_id();
267 if ( is_wp_error( $site_id ) ) {
268 return $site_id;
269 }
270
271 if ( defined( 'IS_WPCOM' ) && IS_WPCOM ) {
272 if ( ! class_exists( 'WPCOM_Gather_Tweetstorm' ) ) {
273 \require_lib( 'gather-tweetstorm' );
274 }
275
276 return WPCOM_Gather_Tweetstorm::gather( $url );
277 }
278
279 $response = Client::wpcom_json_api_request_as_blog(
280 sprintf( '/sites/%d/tweetstorm/gather?url=%s', $site_id, rawurlencode( $url ) ),
281 2,
282 array( 'headers' => array( 'content-type' => 'application/json' ) ),
283 null,
284 'wpcom'
285 );
286 if ( is_wp_error( $response ) ) {
287 return $response;
288 }
289
290 $data = json_decode( wp_remote_retrieve_body( $response ) );
291
292 if ( wp_remote_retrieve_response_code( $response ) >= 400 ) {
293 return new WP_Error( $data->code, $data->message, $data->data );
294 }
295
296 return $data;
297 }
298
299 /**
300 * Parse blocks into an array of tweets.
301 *
302 * @param array $blocks {
303 * An array of blocks, with optional editor-specific information, that need to be parsed into tweets.
304 *
305 * @type array $block A single block, in the form produce by parse_blocks().
306 * @type array $attributes Optional. A list of block attributes and their values from the block editor.
307 * @type string $clientId Optional. The clientId of this block in the block editor.
308 * }
309 * @return array An array of tweets.
310 */
311 public static function parse( $blocks ) {
312 // Reset the tweets array.
313 self::$tweets = array();
314
315 $blocks = self::extract_blocks( $blocks );
316
317 if ( empty( $blocks ) ) {
318 return array();
319 }
320
321 // Initialise the tweets array with an empty tweet, so we don't need to check
322 // if we're creating the first tweet while processing blocks.
323 self::start_new_tweet();
324
325 foreach ( $blocks as $block ) {
326 $block_def = self::get_block_definition( $block['name'] );
327
328 // Grab the most recent tweet.
329 $current_tweet = self::get_current_tweet();
330
331 // Break blocks have no content to add, so we can skip the rest of this loop.
332 if ( 'break' === $block_def['type'] ) {
333 self::save_current_tweet( $current_tweet, $block );
334 continue;
335 }
336
337 // Check if we need to start a new tweet.
338 if ( $current_tweet['finished'] || $block_def['force_new'] ) {
339 self::start_new_tweet();
340 }
341
342 // Process the block.
343 self::add_text_to_tweets( $block );
344 self::add_media_to_tweets( $block );
345 self::add_tweet_to_tweets( $block );
346 self::add_embed_to_tweets( $block );
347 }
348
349 return self::clean_return_tweets();
350 }
351
352 /**
353 * If the passed block name is supported, return the block definition.
354 *
355 * @param string $block_name The registered block name.
356 * @return array|null The block definition, if it's supported.
357 */
358 private static function get_block_definition( $block_name ) {
359 if ( isset( self::$supported_blocks[ $block_name ] ) ) {
360 return self::$supported_blocks[ $block_name ];
361 }
362
363 return null;
364 }
365
366 /**
367 * If the block has any text, process it, and add it to the tweet list.
368 *
369 * @param array $block The block to process.
370 */
371 private static function add_text_to_tweets( $block ) {
372 // This is a text block, is there any text?
373 if ( 0 === strlen( $block['text'] ) ) {
374 return;
375 }
376
377 $block_def = self::get_block_definition( $block['name'] );
378
379 // Grab the most recent tweet, so we can append to that if we can.
380 $current_tweet = self::get_current_tweet();
381
382 // If the entire block can't be fit in this tweet, we need to start a new tweet.
383 if ( $current_tweet['changed'] && ! self::is_valid_tweet( trim( $current_tweet['text'] ) . "\n\n{$block['text']}" ) ) {
384 self::start_new_tweet();
385 }
386
387 // Multiline blocks prioritise splitting by line, but are otherwise identical to
388 // normal text blocks. This means we can treat normal text blocks as being
389 // "multiline", but with a single line.
390 if ( 'multiline' === $block_def['type'] ) {
391 $lines = explode( self::$line_separator, $block['text'] );
392 } else {
393 $lines = array( $block['text'] );
394 }
395 $line_total = count( $lines );
396
397 // Keep track of how many characters from this block we've allocated to tweets.
398 $current_character_count = 0;
399
400 for ( $line_count = 0; $line_count < $line_total; $line_count++ ) {
401 $line_text = $lines[ $line_count ];
402
403 // Make sure we have the most recent tweet at the start of every loop.
404 $current_tweet = self::get_current_tweet();
405
406 if ( $current_tweet['changed'] ) {
407 // When it's the first line, add an extra blank line to seperate
408 // the tweet text from that of the previous block.
409 $separator = "\n\n";
410 if ( $line_count > 0 ) {
411 $separator = "\n";
412 }
413
414 // Is this line short enough to append to the current tweet?
415 if ( self::is_valid_tweet( trim( $current_tweet['text'] ) . "$separator$line_text" ) ) {
416 // Don't trim the text yet, as we may need it for boundary calculations.
417 $current_tweet['text'] = $current_tweet['text'] . "$separator$line_text";
418
419 self::save_current_tweet( $current_tweet, $block );
420 continue;
421 }
422
423 // This line is too long, and lines *must* be split to a new tweet if they don't fit
424 // into the current tweet. If this isn't the first line, record where we split the block.
425 if ( $line_count > 0 ) {
426 // Increment by 1 to allow for the \n between lines to be counted by ::get_boundary().
427 $current_character_count += strlen( $current_tweet['text'] ) + 1;
428 $current_tweet['boundary'] = self::get_boundary( $block, $current_character_count );
429
430 self::save_current_tweet( $current_tweet );
431 }
432
433 // Start a new tweet.
434 $current_tweet = self::start_new_tweet();
435 }
436
437 // Since we're now at the start of a new tweet, is this line short enough to be a tweet by itself?
438 if ( self::is_valid_tweet( $line_text ) ) {
439 $current_tweet['text'] = $line_text;
440
441 self::save_current_tweet( $current_tweet, $block );
442 continue;
443 }
444
445 // The line is too long for a single tweet, so split it by sentences, or linebreaks.
446 $sentences = preg_split( '/(?|(?<!\.\.\.)(?<=[.?!]|\.\)|\.["\'])(\s+)(?=[\p{L}\'"\(])|(\n+))/u', $line_text, -1, PREG_SPLIT_DELIM_CAPTURE );
447 $sentence_total = count( $sentences );
448
449 // preg_split() puts the blank space between sentences into a seperate entry in the result,
450 // so we need to step through the result array by two, and append the blank space when needed.
451 for ( $sentence_count = 0; $sentence_count < $sentence_total; $sentence_count += 2 ) {
452 $current_sentence = $sentences[ $sentence_count ];
453 if ( isset( $sentences[ $sentence_count + 1 ] ) ) {
454 $current_sentence .= $sentences[ $sentence_count + 1 ];
455 }
456
457 // Make sure we have the most recent tweet.
458 $current_tweet = self::get_current_tweet();
459
460 // After the first sentence, we can try and append sentences to the previous sentence.
461 if ( $current_tweet['changed'] && $sentence_count > 0 ) {
462 // Is this sentence short enough for appending to the current tweet?
463 if ( self::is_valid_tweet( $current_tweet['text'] . rtrim( $current_sentence ) ) ) {
464 $current_tweet['text'] .= $current_sentence;
465
466 self::save_current_tweet( $current_tweet, $block );
467 continue;
468 }
469 }
470
471 // Will this sentence fit in its own tweet?
472 if ( self::is_valid_tweet( trim( $current_sentence ) ) ) {
473 if ( $current_tweet['changed'] ) {
474 // If we're already in the middle of a block, record the boundary
475 // before creating a new tweet.
476 if ( $line_count > 0 || $sentence_count > 0 ) {
477 $current_character_count += strlen( $current_tweet['text'] );
478 $current_tweet['boundary'] = self::get_boundary( $block, $current_character_count );
479
480 self::save_current_tweet( $current_tweet );
481 }
482
483 $current_tweet = self::start_new_tweet();
484 }
485 $current_tweet['text'] = $current_sentence;
486
487 self::save_current_tweet( $current_tweet, $block );
488 continue;
489 }
490
491 // This long sentence will start the next tweet that this block is going
492 // to be turned into, so we need to record the boundary and start a new tweet.
493 if ( $current_tweet['changed'] ) {
494 $current_character_count += strlen( $current_tweet['text'] );
495 $current_tweet['boundary'] = self::get_boundary( $block, $current_character_count );
496
497 self::save_current_tweet( $current_tweet );
498
499 $current_tweet = self::start_new_tweet();
500 }
501
502 // Split the long sentence into words.
503 $words = preg_split( '/(\p{Z})/u', $current_sentence, -1, PREG_SPLIT_DELIM_CAPTURE );
504 $word_total = count( $words );
505 for ( $word_count = 0; $word_count < $word_total; $word_count += 2 ) {
506 // Make sure we have the most recent tweet.
507 $current_tweet = self::get_current_tweet();
508
509 // If we're on a new tweet, we don't want to add a space at the start.
510 if ( ! $current_tweet['changed'] ) {
511 $current_tweet['text'] = $words[ $word_count ];
512
513 self::save_current_tweet( $current_tweet, $block );
514 continue;
515 }
516
517 // Can we add this word to the current tweet?
518 if ( self::is_valid_tweet( "{$current_tweet['text']} {$words[ $word_count ]}" ) ) {
519 $space = isset( $words[ $word_count - 1 ] ) ? $words[ $word_count - 1 ] : ' ';
520
521 $current_tweet['text'] .= $space . $words[ $word_count ];
522
523 self::save_current_tweet( $current_tweet, $block );
524 continue;
525 }
526
527 // Add one for the space character that we won't include in the tweet text.
528 $current_character_count += strlen( $current_tweet['text'] ) + 1;
529
530 // We're starting a new tweet with this word. Append ellipsis to
531 // the current tweet, then move on.
532 $current_tweet['text'] .= '';
533
534 $current_tweet['boundary'] = self::get_boundary( $block, $current_character_count );
535 self::save_current_tweet( $current_tweet );
536
537 $current_tweet = self::start_new_tweet();
538
539 // If this is the second tweet created by the split sentence, it'll start
540 // with ellipsis, which we don't want to count, but we do want to count the space
541 // that was replaced by this ellipsis.
542 $current_tweet['text'] = "{$words[ $word_count ]}";
543 $current_character_count -= strlen( '' );
544
545 self::save_current_tweet( $current_tweet, $block );
546 }
547 }
548 }
549 }
550
551 /**
552 * Check if the block has any media to add, and add it.
553 *
554 * @param array $block The block to process.
555 */
556 private static function add_media_to_tweets( $block ) {
557 // There's some media to attach!
558 $media_count = count( $block['media'] );
559 if ( 0 === $media_count ) {
560 return;
561 }
562
563 $current_tweet = self::get_current_tweet();
564
565 // We can only attach media to the previous tweet if the previous tweet
566 // doesn't already have media.
567 if ( count( $current_tweet['media'] ) > 0 ) {
568 $current_tweet = self::start_new_tweet();
569 }
570
571 // Would adding this media make the text of the previous tweet too long?
572 if ( ! self::is_valid_tweet( $current_tweet['text'], $media_count * self::$characters_per_media ) ) {
573 $current_tweet = self::start_new_tweet();
574 }
575
576 $media = array_values(
577 array_filter(
578 $block['media'],
579 function ( $single ) {
580 // Only images and videos can be uploaded.
581 if ( 0 === strpos( $single['type'], 'image/' ) || 0 === strpos( $single['type'], 'video/' ) ) {
582 return true;
583 }
584
585 return false;
586 }
587 )
588 );
589
590 if ( count( $media ) > 0 ) {
591 if ( 0 === strpos( $media[0]['type'], 'video/' ) || 'image/gif' === $media[0]['type'] ) {
592 // We can only attach a single video or GIF.
593 $current_tweet['media'] = array_slice( $media, 0, 1 );
594 } else {
595 // Since a GIF or video isn't the first element, we can remove all of them from the array.
596 $filtered_media = array_values(
597 array_filter(
598 $media,
599 function ( $single ) {
600 if ( 0 === strpos( $single['type'], 'video/' ) || 'image/gif' === $single['type'] ) {
601 return false;
602 }
603
604 return true;
605 }
606 )
607 );
608 // We can only add the first four images found to the tweet.
609 $current_tweet['media'] = array_slice( $filtered_media, 0, 4 );
610 }
611
612 self::save_current_tweet( $current_tweet, $block );
613 }
614 }
615
616 /**
617 * Check if the block has a tweet that we can attach to the current tweet as a quote, and add it.
618 *
619 * @param array $block The block to process.
620 */
621 private static function add_tweet_to_tweets( $block ) {
622 if ( 0 === strlen( $block['tweet'] ) ) {
623 return;
624 }
625
626 $current_tweet = self::get_current_tweet();
627
628 // We can only attach a tweet to the previous tweet if the previous tweet
629 // doesn't already have a tweet quoted.
630 if ( strlen( $current_tweet['tweet'] ) > 0 ) {
631 $current_tweet = self::start_new_tweet();
632 }
633
634 $current_tweet['tweet'] = $block['tweet'];
635
636 self::save_current_tweet( $current_tweet, $block );
637 }
638
639 /**
640 * Check if the block has an embed URL that we can append to the current tweet text.
641 *
642 * @param array $block The block to process.
643 */
644 private static function add_embed_to_tweets( $block ) {
645 if ( 0 === strlen( $block['embed'] ) ) {
646 return;
647 }
648
649 $current_tweet = self::get_current_tweet();
650
651 $reserved_characters = count( $current_tweet['media'] ) * self::$characters_per_media;
652 $reserved_characters += 1 + self::$characters_per_url;
653
654 // We can only attach an embed to the previous tweet if it doesn't already
655 // have any URLs in it. Also, we can't attach it if it'll make the tweet too long.
656 if ( preg_match( '/url-placeholder-\d+-*/', $current_tweet['text'] ) || ! self::is_valid_tweet( $current_tweet['text'], $reserved_characters ) ) {
657 $current_tweet = self::start_new_tweet();
658 $current_tweet['text'] = self::generate_url_placeholder( $block['embed'] );
659 } else {
660 $space = empty( $current_tweet['text'] ) ? '' : ' ';
661 $current_tweet['text'] .= $space . self::generate_url_placeholder( $block['embed'] );
662 }
663
664 self::save_current_tweet( $current_tweet, $block );
665 }
666
667 /**
668 * Given an array of blocks and optional editor information, this will extract them into
669 * the internal representation used during parsing.
670 *
671 * @param array $blocks An array of blocks and optional editor-related information.
672 * @return array An array of blocks, in our internal representation.
673 */
674 private static function extract_blocks( $blocks ) {
675 if ( empty( $blocks ) ) {
676 return array();
677 }
678
679 $block_count = count( $blocks );
680
681 for ( $ii = 0; $ii < $block_count; $ii++ ) {
682 if ( ! self::get_block_definition( $blocks[ $ii ]['block']['blockName'] ) ) {
683 unset( $blocks[ $ii ] );
684 continue;
685 }
686
687 $blocks[ $ii ]['name'] = $blocks[ $ii ]['block']['blockName'];
688 $blocks[ $ii ]['text'] = self::extract_text_from_block( $blocks[ $ii ]['block'] );
689 $blocks[ $ii ]['media'] = self::extract_media_from_block( $blocks[ $ii ]['block'] );
690 $blocks[ $ii ]['tweet'] = self::extract_tweet_from_block( $blocks[ $ii ]['block'] );
691 $blocks[ $ii ]['embed'] = self::extract_embed_from_block( $blocks[ $ii ]['block'] );
692 }
693
694 return array_values( $blocks );
695 }
696
697 /**
698 * Creates a blank tweet, appends it to the tweets array, and returns the tweet.
699 *
700 * @return array The blank tweet.
701 */
702 private static function start_new_tweet() {
703 self::$tweets[] = array(
704 // An array of blocks that make up this tweet.
705 'blocks' => array(),
706 // If this tweet only contains part of a block, the boundary contains
707 // information about where in the block the tweet ends.
708 'boundary' => false,
709 // The text content of the tweet.
710 'text' => '',
711 // The media content of the tweet.
712 'media' => array(),
713 // The quoted tweet in this tweet.
714 'tweet' => '',
715 // Some blocks force a hard finish to the tweet, even if subsequent blocks
716 // could technically be appended. This flag shows when a tweet is finished.
717 'finished' => false,
718 // Flag if the current tweet already has content in it.
719 'changed' => false,
720 );
721
722 return self::get_current_tweet();
723 }
724
725 /**
726 * Get the last tweet in the array.
727 *
728 * @return array The tweet.
729 */
730 private static function get_current_tweet() {
731 return end( self::$tweets );
732 }
733
734 /**
735 * Saves the passed tweet array as the last tweet, overwriting the former last tweet.
736 *
737 * This method adds some last minute checks: marking the tweet as "changed", as well
738 * as adding the $block to the tweet (if it was passed, and hasn't already been added).
739 *
740 * @param array $tweet The tweet being stored.
741 * @param array $block Optional. The block that was used to modify this tweet.
742 * @return array The saved tweet, after the last minute checks have been done.
743 */
744 private static function save_current_tweet( $tweet, $block = null ) {
745 $tweet['changed'] = true;
746
747 if ( isset( $block ) ) {
748 $block_def = self::get_block_definition( $block['name'] );
749
750 // Check if this block type will be forcing a new tweet.
751 if ( $block_def['force_finished'] ) {
752 $tweet['finished'] = true;
753 }
754
755 // Check if this block is already recorded against this tweet.
756 $last_block = end( $tweet['blocks'] );
757 if ( isset( $block['clientId'] ) && ( false === $last_block || $last_block['clientId'] !== $block['clientId'] ) ) {
758 $tweet['blocks'][] = $block;
759 }
760 }
761
762 // Find the index of the last tweet in the array.
763 end( self::$tweets );
764 $tweet_index = key( self::$tweets );
765
766 self::$tweets[ $tweet_index ] = $tweet;
767
768 return $tweet;
769 }
770
771 /**
772 * Checks if the passed text is valid for a tweet or not.
773 *
774 * @param string $text The text to check.
775 * @param int $reserved_characters Optional. The number of characters to reduce the maximum tweet length by.
776 * @return bool Whether or not the text is valid.
777 */
778 private static function is_valid_tweet( $text, $reserved_characters = 0 ) {
779 return self::is_within_twitter_length( $text, 280 - $reserved_characters );
780 }
781
782 /**
783 * Checks if the passed text is valid for image alt text.
784 *
785 * @param string $text The text to check.
786 * @return bool Whether or not the text is valid.
787 */
788 private static function is_valid_alt_text( $text ) {
789 return self::is_within_twitter_length( $text, 1000 );
790 }
791
792 /**
793 * Check if a string is shorter than a given length, according to Twitter's rules for counting string length.
794 *
795 * @param string $text The text to check.
796 * @param int $max_length The number of characters long this string can be.
797 * @return bool Whether or not the string is no longer than the length limit.
798 */
799 private static function is_within_twitter_length( $text, $max_length ) {
800 // Replace all multiline separators with a \n, since that's the
801 // character we actually want to count.
802 $text = str_replace( self::$line_separator, "\n", $text );
803
804 // Keep a running total of characters we've removed.
805 $stripped_characters = 0;
806
807 // Since we use '…' a lot, strip it out, so we can still use the ASCII checks.
808 $ellipsis_count = 0;
809 $text = str_replace( '', '', $text, $ellipsis_count );
810
811 // The ellipsis glyph counts for two characters.
812 $stripped_characters += $ellipsis_count * 2;
813
814 // Try filtering out emoji first, since ASCII text + emoji is a relatively common case.
815 if ( ! self::is_ascii( $text ) ) {
816 // Initialise the emoji cache.
817 if ( 0 === count( self::$emoji_list ) ) {
818 self::$emoji_list = array_map( 'html_entity_decode', _wp_emoji_list( 'entities' ) );
819 }
820
821 $emoji_count = 0;
822 $text = str_replace( self::$emoji_list, '', $text, $emoji_count );
823
824 // Emoji graphemes count as 2 characters each.
825 $stripped_characters += $emoji_count * 2;
826 }
827
828 if ( self::is_ascii( $text ) ) {
829 $stripped_characters += strlen( $text );
830 if ( $stripped_characters <= $max_length ) {
831 return true;
832 }
833
834 return false;
835 }
836
837 // Remove any glyphs that count as 1 character.
838 // Source: https://github.com/twitter/twitter-text/blob/master/config/v3.json .
839 // Note that the source ranges are in decimal, the regex ranges are converted to hex.
840 $single_character_count = 0;
841 $text = preg_replace( '/[\x{0000}-\x{10FF}\x{2000}-\x{200D}\x{2010}-\x{201F}\x{2032}-\x{2037}]/uS', '', $text, -1, $single_character_count );
842
843 $stripped_characters += $single_character_count;
844
845 // Check if there's any text we haven't counted yet.
846 // Any remaining glyphs count as 2 characters each.
847 if ( 0 !== strlen( $text ) ) {
848 // WP provides a compat version of mb_strlen(), no need to check if it exists.
849 $stripped_characters += mb_strlen( $text, 'UTF-8' ) * 2;
850 }
851
852 if ( $stripped_characters <= $max_length ) {
853 return true;
854 }
855
856 return false;
857 }
858
859 /**
860 * Checks if a string only contains ASCII characters.
861 *
862 * @param string $text The string to check.
863 * @return bool Whether or not the string is ASCII-only.
864 */
865 private static function is_ascii( $text ) {
866 if ( function_exists( 'mb_check_encoding' ) ) {
867 if ( mb_check_encoding( $text, 'ASCII' ) ) {
868 return true;
869 }
870 } elseif ( ! preg_match( '/[^\x00-\x7F]/', $text ) ) {
871 return true;
872 }
873
874 return false;
875 }
876
877 /**
878 * A block will generate a certain amount of text to be inserted into a tweet. If that text is too
879 * long for a tweet, we already know where the text will be split when it's published as tweet, but
880 * we need to calculate where that corresponds to in the block edit UI.
881 *
882 * The tweet template for that block may add extra characters, extra characters are added for URL
883 * placeholders, and the block may contain multiple RichText areas (corresponding to attributes),
884 * so we need to keep track of both until the this function calculates which attribute area (in the
885 * block editor, the richTextIdentifier) that offset corresponds to, and how far into that attribute
886 * area it is.
887 *
888 * @param array $block The block being checked.
889 * @param integer $offset The position in the tweet text where it will be split.
890 * @return array|false `false` if the boundary can't be determined. Otherwise, returns the
891 * position in the block editor to insert the tweet boundary annotation.
892 */
893 private static function get_boundary( $block, $offset ) {
894 // If we don't have a clientId, there's no point in generating a boundary, since this
895 // parse request doesn't have a way to map blocks back to editor UI.
896 if ( ! isset( $block['clientId'] ) ) {
897 return false;
898 }
899
900 $block_def = self::get_block_definition( $block['name'] );
901
902 if ( isset( $block_def['content'] ) && count( $block_def['content'] ) > 0 ) {
903 $tags = $block_def['content'];
904 } else {
905 $tags = array( 'content' );
906 }
907
908 $tag_content = self::extract_tag_content_from_html( $tags, $block['block']['innerHTML'] );
909
910 // $tag_content is split up by tag first, then lines. We want to remap it to split it by lines
911 // first, then tag.
912 $lines = array();
913 foreach ( $tag_content as $tag => $content ) {
914 if ( 'content' === $tag ) {
915 $attribute_name = 'content';
916 } else {
917 $attribute_name = array_search( $tag, $block_def['content'], true );
918 }
919
920 foreach ( $content as $id => $content_string ) {
921 // Multiline blocks can have multiple lines, but other blocks will always only have 1.
922 if ( 'multiline' === $block_def['type'] ) {
923 $line_number = $id;
924 } else {
925 $line_number = 0;
926 }
927
928 if ( ! isset( $lines[ $line_number ] ) ) {
929 $lines[ $line_number ] = array();
930 }
931
932 if ( ! isset( $lines[ $line_number ][ $attribute_name ] ) ) {
933 // For multiline blocks, or the first time this attribute has been encountered
934 // in single line blocks, assign the string to the line/attribute.
935 $lines[ $line_number ][ $attribute_name ] = $content_string;
936 } else {
937 // For subsequent times this line/attribute is encountered (only in single line blocks),
938 // append the string with a line break.
939 $lines[ $line_number ][ $attribute_name ] .= "\n$content_string";
940 }
941 }
942 }
943
944 $line_count = count( $lines );
945
946 $template_parts = preg_split( '/({{\w+}})/', $block_def['template'], -1, PREG_SPLIT_DELIM_CAPTURE );
947
948 // Keep track of the total number of bytes we've processed from this block.
949 $total_bytes_processed = 0;
950
951 // Keep track of the number of characters that the processed data translates to in the editor.
952 $characters_processed = 0;
953
954 foreach ( $lines as $line_number => $line ) {
955 // Add up the length of all the parts of this line.
956 $line_byte_total = array_sum( array_map( 'strlen', $line ) );
957
958 if ( $line_byte_total > 0 ) {
959 // We have something to use in the template, so loop over each part of the template, and count it.
960 foreach ( $template_parts as $template_part ) {
961 $matches = array();
962 if ( preg_match( '/{{(\w+)}}/', $template_part, $matches ) ) {
963 $part_name = $matches[1];
964
965 $line_part_data = $line[ $part_name ];
966 $line_part_bytes = strlen( $line_part_data );
967
968 $cleaned_line_part_data = preg_replace( '/ \(url-placeholder-\d+-*\)/', '', $line_part_data );
969
970 $cleaned_line_part_data = preg_replace_callback(
971 '/url-placeholder-(\d+)-*/',
972 function ( $matches ) {
973 return self::$urls[ $matches[1] ];
974 },
975 $cleaned_line_part_data
976 );
977
978 if ( $total_bytes_processed + $line_part_bytes >= $offset ) {
979 // We know that the offset is somewhere inside this part of the tweet, but we need to remove the length
980 // of any URL placeholders that appear before the boundary, to be able to calculate the correct attribute offset.
981
982 // $total_bytes_processed is the sum of everything we've processed so far, (including previous parts)
983 // on this line. This makes it relatively easy to calculate the number of bytes into this part
984 // that the boundary will occur.
985 $line_part_byte_boundary = $offset - $total_bytes_processed;
986
987 // Grab the data from this line part that appears before the boundary.
988 $line_part_pre_boundary_data = substr( $line_part_data, 0, $line_part_byte_boundary );
989
990 // Remove any URL placeholders, since these aren't shown in the editor.
991 $line_part_pre_boundary_data = preg_replace( '/ \(url-placeholder-\d+-*\)/', '', $line_part_pre_boundary_data );
992
993 $line_part_pre_boundary_data = preg_replace_callback(
994 '/url-placeholder-(\d+)-*/',
995 function ( $matches ) {
996 return self::$urls[ $matches[1] ];
997 },
998 $line_part_pre_boundary_data
999 );
1000
1001 $boundary_start = self::utf_16_code_unit_length( $line_part_pre_boundary_data ) - 1;
1002
1003 // Multiline blocks need to offset for the characters that are in the same content area,
1004 // but which were counted on previous lines.
1005 if ( 'multiline' === $block_def['type'] ) {
1006 $boundary_start += $characters_processed;
1007 }
1008
1009 // Check if the boundary is happening on a line break or a space.
1010 if ( "\n" === $line_part_data[ $line_part_byte_boundary - 1 ] ) {
1011 $type = 'line-break';
1012
1013 // A line break boundary can actually be multiple consecutive line breaks,
1014 // count them all up so we know how big the annotation needs to be.
1015 $matches = array();
1016 preg_match( '/\n+$/', substr( $line_part_data, 0, $line_part_byte_boundary ), $matches );
1017 $boundary_end = $boundary_start + 1;
1018 $boundary_start -= strlen( $matches[0] ) - 1;
1019 } else {
1020 $type = 'normal';
1021 $boundary_end = $boundary_start + 1;
1022 }
1023
1024 return array(
1025 'start' => $boundary_start,
1026 'end' => $boundary_end,
1027 'container' => $part_name,
1028 'type' => $type,
1029 );
1030 } else {
1031 $total_bytes_processed += $line_part_bytes;
1032 $characters_processed += self::utf_16_code_unit_length( $cleaned_line_part_data );
1033 continue;
1034 }
1035 } else {
1036 $total_bytes_processed += strlen( $template_part );
1037 }
1038 }
1039
1040 // Are we breaking at the end of this line?
1041 if ( $total_bytes_processed + 1 === $offset && $line_count > 1 ) {
1042 reset( $block_def['content'] );
1043 $container = key( $block_def['content'] );
1044 return array(
1045 'line' => $line_number,
1046 'container' => $container,
1047 'type' => 'end-of-line',
1048 );
1049 }
1050
1051 // The newline at the end of each line is 1 byte, but we don't need to count empty lines.
1052 ++$total_bytes_processed;
1053 }
1054
1055 // We do need to count empty lines in the editor, since they'll be displayed.
1056 ++$characters_processed;
1057 }
1058
1059 return false;
1060 }
1061
1062 /**
1063 * JavaScript uses UTF-16 for encoding strings, which means we need to provide UTF-16
1064 * based offsets for the block editor to render tweet boundaries in the correct location.
1065 *
1066 * UTF-16 is a variable-width character encoding: every code unit is 2 bytes, a single character
1067 * can be one or two code units long. Fortunately for us, JavaScript's String.charAt() is based
1068 * on the older UCS-2 character encoding, which only counts single code units. PHP's strlen()
1069 * counts a code unit as being 2 characters, so once a string is converted to UTF-16, we have
1070 * a fast way to determine how long it is in UTF-16 code units.
1071 *
1072 * @param string $text The natively encoded string to get the length of.
1073 * @return int The length of the string in UTF-16 code units. Returns -1 if the length could not
1074 * be calculated.
1075 */
1076 private static function utf_16_code_unit_length( $text ) {
1077 // If mb_convert_encoding() exists, we can use that for conversion.
1078 if ( function_exists( 'mb_convert_encoding' ) ) {
1079 // UTF-16 can add an additional code unit to the start of the string, called the
1080 // Byte Order Mark (BOM), which indicates whether the string is encoding as
1081 // big-endian, or little-endian. Since we don't want to count code unit, and the endianness
1082 // doesn't matter for our purposes, using PHP's UTF-16BE encoding uses big-endian
1083 // encoding, and ensures the BOM *won't* be prepended to the string to the string.
1084 return strlen( mb_convert_encoding( $text, 'UTF-16BE' ) ) / 2;
1085 }
1086
1087 // If we can't convert this string, return a result that will avoid an incorrect annotation being added.
1088 return -1;
1089 }
1090
1091 /**
1092 * Extracts the tweetable text from a block.
1093 *
1094 * @param array $block A single block, as generated by parse_block().
1095 * @return string The tweetable text from the block, in the correct template form.
1096 */
1097 private static function extract_text_from_block( $block ) {
1098 // If the block doesn't have an innerHTMl, we're not going to get any text.
1099 if ( empty( $block['innerHTML'] ) ) {
1100 return '';
1101 }
1102
1103 $block_def = self::get_block_definition( $block['blockName'] );
1104
1105 // We currently only support extracting text from HTML text nodes.
1106 if ( ! isset( $block_def['content_location'] ) || 'html' !== $block_def['content_location'] ) {
1107 return '';
1108 }
1109
1110 // Find out which tags we need to extract content from.
1111 if ( isset( $block_def['content'] ) && count( $block_def['content'] ) > 0 ) {
1112 $tags = $block_def['content'];
1113 } else {
1114 $tags = array( 'content' );
1115 }
1116
1117 $tag_values = self::extract_tag_content_from_html( $tags, $block['innerHTML'] );
1118
1119 // We can treat single line blocks as "multiline", with only one line in them.
1120 $lines = array();
1121 foreach ( $tag_values as $tag => $values ) {
1122 // For single-line blocks, we need to squash all the values for this tag into a single value.
1123 if ( 'multiline' !== $block_def['type'] ) {
1124 $values = array( implode( "\n", $values ) );
1125 }
1126
1127 // Handle the special "content" tag.
1128 if ( 'content' === $tag ) {
1129 $placeholder = 'content';
1130 } else {
1131 $placeholder = array_search( $tag, $block_def['content'], true );
1132 }
1133
1134 // Loop over each instance of this value, appling that value to the corresponding line template.
1135 foreach ( $values as $line_number => $value ) {
1136 if ( ! isset( $lines[ $line_number ] ) ) {
1137 $lines[ $line_number ] = $block_def['template'];
1138 }
1139
1140 $lines[ $line_number ] = str_replace( '{{' . $placeholder . '}}', $value, $lines[ $line_number ] );
1141 }
1142 }
1143
1144 // Remove any lines that didn't apply any content.
1145 $empty_template = preg_replace( '/{{.*?}}/', '', $block_def['template'] );
1146 $lines = array_filter(
1147 $lines,
1148 function ( $line ) use ( $empty_template ) {
1149 return $line !== $empty_template;
1150 }
1151 );
1152
1153 // Join the lines together into a single string.
1154 $text = implode( self::$line_separator, $lines );
1155
1156 // Trim off any trailing whitespace that we no longer need.
1157 $text = preg_replace( '/(\s|' . self::$line_separator . ')+$/u', '', $text );
1158
1159 return $text;
1160 }
1161
1162 /**
1163 * Extracts the tweetable media from a block.
1164 *
1165 * @param array $block A single block, as generated by parse_block().
1166 * @return array {
1167 * An array of media.
1168 *
1169 * @type string url The URL of the media.
1170 * @type string alt The alt text of the media.
1171 * }
1172 */
1173 private static function extract_media_from_block( $block ) {
1174 $block_def = self::get_block_definition( $block['blockName'] );
1175
1176 $media = array();
1177
1178 if ( 'image' === $block_def['type'] ) {
1179 $url = self::extract_attr_content_from_html(
1180 $block_def['content']['url'][0],
1181 $block_def['content']['url'][1],
1182 $block['innerHTML']
1183 );
1184 $alt = self::extract_attr_content_from_html(
1185 $block_def['content']['alt'][0],
1186 $block_def['content']['alt'][1],
1187 $block['innerHTML']
1188 );
1189
1190 $img_count = count( $url );
1191
1192 for ( $ii = 0; $ii < $img_count; $ii++ ) {
1193 $filedata = wp_check_filetype( basename( wp_parse_url( $url[ $ii ], PHP_URL_PATH ) ) );
1194
1195 $media[] = array(
1196 'url' => $url[ $ii ],
1197 'alt' => self::is_valid_alt_text( $alt[ $ii ] ) ? $alt[ $ii ] : '',
1198 'type' => $filedata['type'],
1199 );
1200 }
1201 } elseif ( 'video' === $block_def['type'] ) {
1202 // Handle VideoPress videos.
1203 if ( isset( $block['attrs']['src'] ) && 0 === strpos( $block['attrs']['src'], 'https://videos.files.wordpress.com/' ) ) {
1204 $url = array( $block['attrs']['src'] );
1205 } else {
1206 $url = self::extract_attr_content_from_html(
1207 $block_def['content']['url'][0],
1208 $block_def['content']['url'][1],
1209 $block['innerHTML']
1210 );
1211 }
1212
1213 // We can only ever use the first video found, no need to go through all of them.
1214 if ( count( $url ) > 0 ) {
1215 $filedata = wp_check_filetype( basename( wp_parse_url( $url[0], PHP_URL_PATH ) ) );
1216
1217 $media[] = array(
1218 'url' => $url[0],
1219 'type' => $filedata['type'],
1220 );
1221 }
1222 }
1223
1224 return $media;
1225 }
1226
1227 /**
1228 * Extracts the tweet URL from a Twitter embed block.
1229 *
1230 * @param array $block A single block, as generated by parse_block().
1231 * @return string The tweet URL. Empty string if there is none available.
1232 */
1233 private static function extract_tweet_from_block( $block ) {
1234 if (
1235 'core/embed' === $block['blockName']
1236 && ( isset( $block['attrs']['providerNameSlug'] ) && 'twitter' === $block['attrs']['providerNameSlug'] )
1237 ) {
1238 return $block['attrs']['url'];
1239 }
1240
1241 return '';
1242 }
1243
1244 /**
1245 * Extracts URL from an embed block.
1246 *
1247 * @param array $block A single block, as generated by parse_block().
1248 * @return string The URL. Empty string if there is none available.
1249 */
1250 private static function extract_embed_from_block( $block ) {
1251 $block_def = self::get_block_definition( $block['blockName'] );
1252
1253 if ( 'embed' !== $block_def['type'] ) {
1254 return '';
1255 }
1256
1257 // Twitter embeds are handled in ::extract_tweet_from_block().
1258 if (
1259 'core/embed' === $block['blockName']
1260 && ( isset( $block['attrs']['providerNameSlug'] ) && 'twitter' === $block['attrs']['providerNameSlug'] )
1261 ) {
1262 return '';
1263 }
1264
1265 $url = '';
1266 if ( 'block-attributes' === $block_def['content_location'] ) {
1267 $url = $block['attrs'][ $block_def['content'][0] ];
1268 }
1269
1270 if ( 'jetpack/gif' === $block['blockName'] ) {
1271 $url = str_replace( '/embed/', '/gifs/', $url );
1272 }
1273
1274 return $url;
1275 }
1276
1277 /**
1278 * There's a bunch of left-over cruft in the tweets array that we don't need to return. Removing
1279 * it helps keep the size of the data down.
1280 */
1281 private static function clean_return_tweets() {
1282 // Before we return, clean out unnecessary cruft from the return data.
1283 $tweets = array_map(
1284 function ( $tweet ) {
1285 // Remove tweets that don't have anything saved in them. eg, if the last block is a
1286 // header with no text, it'll force a new tweet, but we won't end up putting anything
1287 // in that tweet.
1288 if ( ! $tweet['changed'] ) {
1289 return false;
1290 }
1291
1292 // Replace any URL placeholders that appear in the text.
1293 $tweet['urls'] = array();
1294 foreach ( self::$urls as $id => $url ) {
1295 $count = 0;
1296
1297 $tweet['text'] = str_replace( str_pad( "url-placeholder-$id", self::$characters_per_url, '-' ), $url, $tweet['text'], $count );
1298
1299 // If we found a URL, keep track of it for the editor.
1300 if ( $count > 0 ) {
1301 $tweet['urls'][] = $url;
1302 }
1303 }
1304
1305 // Remove any inline placeholders.
1306 $tweet['text'] = str_replace( self::$inline_placeholder, '', $tweet['text'] );
1307
1308 // If the tweet text consists only of whitespace, we can remove all of it.
1309 if ( preg_match( '/^\s*$/u', $tweet['text'] ) ) {
1310 $tweet['text'] = '';
1311 }
1312
1313 // Remove trailing whitespace from every line.
1314 $tweet['text'] = preg_replace( '/\p{Z}+$/um', '', $tweet['text'] );
1315
1316 // Remove all trailing whitespace (including line breaks) from the end of the text.
1317 $tweet['text'] = rtrim( $tweet['text'] );
1318
1319 // Remove internal flags.
1320 unset( $tweet['changed'] );
1321 unset( $tweet['finished'] );
1322
1323 // Remove bulky block data.
1324 if ( ! isset( $tweet['blocks'][0]['attributes'] ) && ! isset( $tweet['blocks'][0]['clientId'] ) ) {
1325 $tweet['blocks'] = array();
1326 } else {
1327 // Remove the parts of the block data that the editor doesn't need.
1328 $block_count = count( $tweet['blocks'] );
1329 for ( $ii = 0; $ii < $block_count; $ii++ ) {
1330 $keys = array_keys( $tweet['blocks'][ $ii ] );
1331 foreach ( $keys as $key ) {
1332 // The editor only needs these attributes, everything else will be unset.
1333 if ( in_array( $key, array( 'attributes', 'clientId' ), true ) ) {
1334 continue;
1335 }
1336
1337 unset( $tweet['blocks'][ $ii ][ $key ] );
1338 }
1339 }
1340 }
1341
1342 // Once we've finished cleaning up, check if there's anything left to be tweeted.
1343 if ( empty( $tweet['text'] ) && empty( $tweet['media'] ) && empty( $tweet['tweet'] ) ) {
1344 return false;
1345 }
1346
1347 return $tweet;
1348 },
1349 self::$tweets
1350 );
1351
1352 // Clean any removed tweets out of the result.
1353 return array_values( array_filter( $tweets, 'is_array' ) );
1354 }
1355
1356 /**
1357 * Given a list of tags and a HTML blob, this will extract the text content inside
1358 * each of the given tags.
1359 *
1360 * @param array $tags An array of tag names.
1361 * @param string $html A blob of HTML.
1362 * @return array An array of the extract content. The keys in the array are the $tags,
1363 * each value is an array. The value array is indexed in the same order as the tag
1364 * appears in the HTML blob, including nested tags.
1365 */
1366 private static function extract_tag_content_from_html( $tags, $html ) {
1367 // Serialised blocks will sometimes wrap the innerHTML in newlines, but those newlines
1368 // are removed when innerHTML is parsed into an attribute. Remove them so we're working
1369 // with the same information.
1370 if ( "\n" === $html[0] && "\n" === $html[ strlen( $html ) - 1 ] ) {
1371 $html = substr( $html, 1, strlen( $html ) - 2 );
1372 }
1373
1374 // Normalise <br>.
1375 $html = preg_replace( '/<br\s*\/?>/', '<br>', $html );
1376
1377 // If there were no tags passed, assume the entire text is required.
1378 if ( empty( $tags ) ) {
1379 $tags = array( 'content' );
1380 }
1381
1382 $values = array();
1383
1384 $tokens = wp_html_split( $html );
1385
1386 $validator = new Twitter_Validator();
1387
1388 foreach ( $tags as $tag ) {
1389 $values[ $tag ] = array();
1390
1391 // Since tags can be nested, keeping track of the nesting level allows
1392 // us to extract nested content into a flat array.
1393 if ( 'content' === $tag ) {
1394 // The special "content" tag means we should store the entire content,
1395 // so assume the tag is open from the beginning.
1396 $opened = 0;
1397 $closed = -1;
1398
1399 $values['content'][0] = '';
1400 } else {
1401 $opened = -1;
1402 $closed = -1;
1403 }
1404
1405 // When we come across a URL, we need to keep track of it, so it can then be inserted
1406 // in the right place.
1407 $current_url = '';
1408 foreach ( $tokens as $token ) {
1409 if ( 0 === strlen( $token ) ) {
1410 // Skip any empty tokens.
1411 continue;
1412 }
1413
1414 // If we're currently storing content, check if it's a text-formatting
1415 // tag that we should apply.
1416 if ( $opened !== $closed ) {
1417 // End of a paragraph, put in some newlines (as long as we're not extracting paragraphs).
1418 if ( '</p>' === $token && 'p' !== $tag ) {
1419 $values[ $tag ][ $opened ] .= "\n\n";
1420 }
1421
1422 // A line break gets one newline.
1423 if ( '<br>' === $token ) {
1424 $values[ $tag ][ $opened ] .= "\n";
1425 }
1426
1427 // A link has opened, grab the URL for inserting later.
1428 if ( 0 === strpos( $token, '<a ' ) ) {
1429 $href_values = self::extract_attr_content_from_html( 'a', 'href', $token );
1430 if ( ! empty( $href_values[0] ) && $validator->isValidURL( $href_values[0] ) ) {
1431 // Remember the URL.
1432 $current_url = $href_values[0];
1433 }
1434 }
1435
1436 // A link has closed, insert the URL from that link if we have one.
1437 if ( '</a>' === $token && '' !== $current_url ) {
1438 // Generate a unique-to-this-block placeholder which takes up the
1439 // same number of characters as a URL does.
1440 $values[ $tag ][ $opened ] .= ' (' . self::generate_url_placeholder( $current_url ) . ')';
1441
1442 $current_url = '';
1443 }
1444
1445 // We don't return inline images, but they technically take up 1 character in the RichText.
1446 if ( 0 === strpos( $token, '<img ' ) ) {
1447 $values[ $tag ][ $opened ] .= self::$inline_placeholder;
1448 }
1449 }
1450
1451 if ( "<$tag>" === $token || 0 === strpos( $token, "<$tag " ) ) {
1452 // A tag has just been opened.
1453 ++$opened;
1454 // Set an empty value now, so we're keeping track of empty tags.
1455 if ( ! isset( $values[ $tag ][ $opened ] ) ) {
1456 $values[ $tag ][ $opened ] = '';
1457 }
1458 continue;
1459 }
1460
1461 if ( "</$tag>" === $token ) {
1462 // The tag has been closed.
1463 ++$closed;
1464 continue;
1465 }
1466
1467 if ( '<' === $token[0] ) {
1468 // We can skip any other tags.
1469 continue;
1470 }
1471
1472 if ( $opened !== $closed ) {
1473 // We're currently in a tag, with some content. Start by decoding any HTML entities.
1474 $token = html_entity_decode( $token, ENT_QUOTES );
1475
1476 // Find any URLs in this content, and replace them with a placeholder.
1477 preg_match_all( Twitter_Regex::getValidUrlMatcher(), $token, $matches, PREG_SET_ORDER | PREG_OFFSET_CAPTURE );
1478 $offset = 0;
1479 foreach ( $matches as $match ) {
1480 list( $url, $start ) = $match[2];
1481
1482 $token = substr_replace( $token, self::generate_url_placeholder( $url ), $start + $offset, strlen( $url ) );
1483
1484 $offset += self::$characters_per_url - strlen( $url );
1485
1486 // If we're in a link with a URL set, there's no need to keep two copies of the same link.
1487 if ( ! empty( $current_url ) ) {
1488 $lower_url = strtolower( $url );
1489 $lower_current_url = strtolower( $current_url );
1490
1491 if ( $lower_url === $lower_current_url ) {
1492 $current_url = '';
1493 }
1494
1495 // Check that the link text isn't just a shortened version of the href value.
1496 $trimmed_current_url = preg_replace( '|^https?://|', '', $lower_current_url );
1497 if ( $lower_url === $trimmed_current_url || trim( $trimmed_current_url, '/' ) === $lower_url ) {
1498 $current_url = '';
1499 }
1500 }
1501 }
1502
1503 // Append it to the right value.
1504 $values[ $tag ][ $opened ] .= $token;
1505 }
1506 }
1507 }
1508
1509 return $values;
1510 }
1511
1512 /**
1513 * Extracts the attribute content from a tag.
1514 *
1515 * This method allows for the HTML to have multiple instances of the tag, and will return
1516 * an array containing the attribute value (or an empty string, if the tag doesn't have the
1517 * requested attribute) for each occurrence of the tag.
1518 *
1519 * @param string $tag The tag we're looking for.
1520 * @param string $attr The name of the attribute we're looking for.
1521 * @param string $html The HTML we're searching through.
1522 * @param array $attr_filters Optional. Filters tags based on whether or not they have attributes with given values.
1523 * @return array The array of attribute values found.
1524 */
1525 private static function extract_attr_content_from_html( $tag, $attr, $html, $attr_filters = array() ) {
1526 // Given our single tag and attribute, construct a KSES filter for it.
1527 $kses_filter = array(
1528 $tag => array(
1529 $attr => array(),
1530 ),
1531 );
1532
1533 foreach ( $attr_filters as $filter_attr => $filter_value ) {
1534 $kses_filter[ $tag ][ $filter_attr ] = array();
1535 }
1536
1537 // Remove all HTML except for the tag we're after. On that tag,
1538 // remove all attributes except for the one we're after.
1539 $stripped_html = wp_kses( $html, $kses_filter );
1540
1541 $values = array();
1542
1543 $tokens = wp_html_split( $stripped_html );
1544 foreach ( $tokens as $token ) {
1545 $found_value = '';
1546
1547 if ( 0 === strlen( $token ) ) {
1548 // Skip any empty tokens.
1549 continue;
1550 }
1551
1552 if ( '<' !== $token[0] ) {
1553 // We can skip any non-tag tokens.
1554 continue;
1555 }
1556
1557 $token_attrs = wp_kses_attr_parse( $token );
1558
1559 // Skip tags that KSES couldn't handle.
1560 if ( false === $token_attrs ) {
1561 continue;
1562 }
1563
1564 // Remove the tag open and close chunks.
1565 $found_tag = array_shift( $token_attrs );
1566 array_pop( $token_attrs );
1567
1568 // We somehow got a tag that isn't the one we're after. Skip it.
1569 if ( 0 !== strpos( $found_tag, "<$tag " ) ) {
1570 continue;
1571 }
1572
1573 // We can only fail an attribute filter if one is set.
1574 $passed_filter = count( $attr_filters ) === 0;
1575
1576 foreach ( $token_attrs as $token_attr_string ) {
1577 // The first "=" in the string will be between the attribute name/value.
1578 list( $token_attr_name, $token_attr_value ) = explode( '=', $token_attr_string, 2 );
1579
1580 $token_attr_name = trim( $token_attr_name );
1581 $token_attr_value = trim( $token_attr_value );
1582
1583 // Remove a single set of quotes from around the value.
1584 if ( '' !== $token_attr_value && in_array( $token_attr_value[0], array( '"', "'" ), true ) ) {
1585 $token_attr_value = trim( $token_attr_value, $token_attr_value[0] );
1586 }
1587
1588 // If this is the attribute we're after, save the value for the end of the loop.
1589 if ( $token_attr_name === $attr ) {
1590 $found_value = $token_attr_value;
1591 }
1592
1593 if ( isset( $attr_filters[ $token_attr_name ] ) && $attr_filters[ $token_attr_name ] === $token_attr_value ) {
1594 $passed_filter = true;
1595 }
1596 }
1597
1598 if ( $passed_filter ) {
1599 // We always want to append the found value, even if we didn't "find" a matching attribute.
1600 // An empty string in the return value means that we found the tag, but the attribute was
1601 // either empty, or not set.
1602 $values[] = html_entity_decode( $found_value, ENT_QUOTES );
1603 }
1604 }
1605
1606 return $values;
1607 }
1608
1609 /**
1610 * Generates a placeholder for URLs, using the appropriate number of characters to imitate how
1611 * Twitter counts the length of URLs in tweets.
1612 *
1613 * @param string $url The URL to generate a placeholder for.
1614 * @return string The placeholder.
1615 */
1616 public static function generate_url_placeholder( $url ) {
1617 self::$urls[] = $url;
1618
1619 return str_pad( 'url-placeholder-' . ( count( self::$urls ) - 1 ), self::$characters_per_url, '-' );
1620 }
1621
1622 /**
1623 * Retrieves the Twitter card data for a list of URLs.
1624 *
1625 * @param array $urls The list of URLs to grab Twitter card data for.
1626 * @return array The Twitter card data.
1627 */
1628 public static function generate_cards( $urls ) {
1629 global $wp_version;
1630
1631 $validator = new Twitter_Validator();
1632
1633 $requests = array_map(
1634 function ( $url ) use ( $validator ) {
1635 if (
1636 false !== wp_http_validate_url( $url )
1637 && $validator->isValidURL( $url )
1638 ) {
1639 return array(
1640 'url' => $url,
1641 );
1642 }
1643
1644 return false;
1645 },
1646 $urls
1647 );
1648
1649 $requests = array_filter( $requests );
1650
1651 // Remove this check once WordPress 6.2 is the minimum supported version.
1652 if ( version_compare( $wp_version, '6.2-alpha', '<' ) ) {
1653 $hooks = new Requests_Hooks();
1654 } else {
1655 $hooks = new \WpOrg\Requests\Hooks();
1656 }
1657
1658 $hooks->register(
1659 'requests.before_redirect',
1660 array( self::class, 'validate_redirect_url' )
1661 );
1662
1663 // Remove this check once WordPress 6.2 is the minimum supported version.
1664 $results = version_compare( $wp_version, '6.2-alpha', '<' )
1665 ? Requests::request_multiple( $requests, array( 'hooks' => $hooks ) )
1666 : \WpOrg\Requests\Requests::request_multiple( $requests, array( 'hooks' => $hooks ) );
1667
1668 foreach ( $results as $result ) {
1669 if ( $result instanceof Requests_Exception || $result instanceof \WpOrg\Requests\Exception ) {
1670 return new WP_Error(
1671 'invalid_url',
1672 __( 'Sorry, something is wrong with the requested URL.', 'jetpack' ),
1673 403
1674 );
1675 }
1676 }
1677
1678 $card_data = array(
1679 'creator' => array(
1680 'name' => 'twitter:creator',
1681 ),
1682 'description' => array(
1683 'name' => 'twitter:description',
1684 'property' => 'og:description',
1685 ),
1686 'image' => array(
1687 'name' => 'twitter:image',
1688 'property' => 'og:image',
1689 ),
1690 'title' => array(
1691 'name' => 'twitter:text:title',
1692 'property' => 'og:title',
1693 ),
1694 'type' => array(
1695 'name' => 'twitter:card',
1696 ),
1697 );
1698
1699 $cards = array();
1700 foreach ( $results as $id => $result ) {
1701 $url = $requests[ $id ]['url'];
1702
1703 if ( ! $result->success ) {
1704 $cards[ $url ] = array(
1705 'error' => 'invalid_url',
1706 );
1707 continue;
1708 }
1709
1710 $url_card_data = array();
1711
1712 foreach ( $card_data as $key => $filters ) {
1713 foreach ( $filters as $attribute => $value ) {
1714 $found_data = self::extract_attr_content_from_html( 'meta', 'content', $result->body, array( $attribute => $value ) );
1715 if ( count( $found_data ) > 0 && strlen( $found_data[0] ) > 0 ) {
1716 $url_card_data[ $key ] = html_entity_decode( $found_data[0], ENT_QUOTES );
1717 break;
1718 }
1719 }
1720 }
1721
1722 if ( count( $url_card_data ) > 0 ) {
1723 $cards[ $url ] = $url_card_data;
1724 } else {
1725 $cards[ $url ] = array(
1726 'error' => 'no_og_data',
1727 );
1728 }
1729 }
1730
1731 return $cards;
1732 }
1733
1734 /**
1735 * Filters the redirect URLs that can appear when requesting passed URLs.
1736 *
1737 * @param String $redirect_url the URL to which a redirect is requested.
1738 * @throws Requests_Exception In case the URL is not validated, if WP version is less than 6.2.
1739 * @throws \WpOrg\Requests\Exception In case the URL is not validated, if WP version is 6.2 or greater.
1740 * @return void
1741 */
1742 public static function validate_redirect_url( $redirect_url ) {
1743 global $wp_version;
1744
1745 if ( ! wp_http_validate_url( $redirect_url ) ) {
1746 // Remove this check once WordPress 6.2 is the minimum supported version.
1747 if ( version_compare( $wp_version, '6.2-alpha', '<' ) ) {
1748 throw new Requests_Exception( __( 'A valid URL was not provided.', 'jetpack' ), 'wp_http.redirect_failed_validation' );
1749 }
1750 throw new \WpOrg\Requests\Exception( __( 'A valid URL was not provided.', 'jetpack' ), 'wp_http.redirect_failed_validation' );
1751 }
1752 }
1753 }
1754