PluginProbe
BetterDocs – AI Documentation, Knowledge Base, MCP Server, Docs, Wikis, FAQ & Chatbot / 4.9.3
BetterDocs – AI Documentation, Knowledge Base, MCP Server, Docs, Wikis, FAQ & Chatbot v4.9.3
4.9.3 4.9.2 4.9.1 4.9.0 4.8.2 4.8.1 4.8.0 4.7.0 4.6.2 4.6.1 4.6.0 4.5.6 4.5.5 4.5.4 4.5.3 4.5.2 4.5.1 4.5.0 4.4.1 4.4.0 3.3.4 3.4.0 3.4.1 3.4.2 3.5.0 All 201 releases
betterdocs / includes / Admin / Importer / Parsers / WXR_Parser_SimpleXML.php

WXR_Parser_SimpleXML.php in BetterDocs – AI Documentation, Knowledge Base, MCP Server, Docs, Wikis, FAQ & Chatbot 4.9.3, at includes/Admin/Importer/Parsers/WXR_Parser_SimpleXML.php

285 lines 9.0 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 namespace WPDeveloper\BetterDocs\Admin\Importer\Parsers;
3
4 use WP_Error;
5
6 if ( ! defined( 'ABSPATH' ) ) {
7 exit; // Exit if accessed directly.
8 }
9
10 /**
11 * WordPress extended RSS file parser implementations,
12 * Originally made by WordPress part of WordPress/Importer.
13 * https://plugins.trac.wordpress.org/browser/wordpress-importer/trunk/parsers/class-wxr-parser-simplexml.php
14 *
15 * What was done (by Elementor):
16 * Reformat of the code.
17 * Removed variable '$internal_errors'.
18 * Changed text domain.
19 */
20
21 /**
22 * WXR Parser that makes use of the SimpleXML PHP extension.
23 */
24 class WXR_Parser_SimpleXML {
25
26 /**
27 * @param string $file
28 *
29 * @return array|WP_Error
30 */
31 public function parse( $file ) {
32 $authors = [];
33 $posts = [];
34 $categories = [];
35 $tags = [];
36 $terms = [];
37
38 libxml_use_internal_errors( true );
39
40 $dom = new \DOMDocument();
41 $old_value = null;
42
43 $libxml_disable_entity_loader_exists = function_exists( 'libxml_disable_entity_loader' );
44
45 if ( $libxml_disable_entity_loader_exists ) {
46 $old_value = libxml_disable_entity_loader( true ); // phpcs:ignore Generic.PHP.DeprecatedFunctions.Deprecated
47 }
48
49 $success = $dom->loadXML( $this->file_get_contents( $file ) );
50
51 if ( $libxml_disable_entity_loader_exists && ! is_null( $old_value ) ) {
52 libxml_disable_entity_loader( $old_value ); // phpcs:ignore Generic.PHP.DeprecatedFunctions.Deprecated
53 }
54
55 if ( ! $success || isset( $dom->doctype ) ) {
56 return new WP_Error( 'SimpleXML_parse_error', esc_html__( 'There was an error when reading this WXR file', 'betterdocs' ), libxml_get_errors() );
57 }
58
59 $xml = simplexml_import_dom( $dom );
60 unset( $dom );
61
62 // Halt if loading produces an error.
63 if ( ! $xml ) {
64 return new WP_Error( 'SimpleXML_parse_error', esc_html__( 'There was an error when reading this WXR file', 'betterdocs' ), libxml_get_errors() );
65 }
66
67 $wxr_version = $xml->xpath( '/rss/channel/wp:wxr_version' );
68 if ( ! $wxr_version ) {
69 return new WP_Error( 'WXR_parse_error', esc_html__( 'This does not appear to be a WXR file, missing/invalid WXR version number', 'betterdocs' ) );
70 }
71
72 $wxr_version = (string) trim( $wxr_version[0] );
73 // Confirm that we are dealing with the correct file format.
74 if ( ! preg_match( '/^\d+\.\d+$/', $wxr_version ) ) {
75 return new WP_Error( 'WXR_parse_error', esc_html__( 'This does not appear to be a WXR file, missing/invalid WXR version number', 'betterdocs' ) );
76 }
77
78 $base_url = $xml->xpath( '/rss/channel/wp:base_site_url' );
79 $base_url = (string) trim( $base_url[0] ?? '' );
80
81 $base_blog_url = $xml->xpath( '/rss/channel/wp:base_blog_url' );
82 if ( $base_blog_url ) {
83 $base_blog_url = (string) trim( $base_blog_url[0] );
84 } else {
85 $base_blog_url = $base_url;
86 }
87
88 $page_on_front = $xml->xpath( '/rss/channel/wp:page_on_front' );
89
90 if ( $page_on_front ) {
91 $page_on_front = (int) $page_on_front[0];
92 }
93
94 $namespaces = $xml->getDocNamespaces();
95 if ( ! isset( $namespaces['wp'] ) ) {
96 $namespaces['wp'] = 'http://wordpress.org/export/1.1/';
97 }
98 if ( ! isset( $namespaces['excerpt'] ) ) {
99 $namespaces['excerpt'] = 'http://wordpress.org/export/1.1/excerpt/';
100 }
101
102 // Grab authors.
103 foreach ( $xml->xpath( '/rss/channel/wp:author' ) as $author_arr ) {
104 $a = $author_arr->children( $namespaces['wp'] );
105 $login = (string) $a->author_login;
106 $authors[ $login ] = [
107 'author_id' => (int) $a->author_id,
108 'author_login' => $login,
109 'author_email' => (string) $a->author_email,
110 'author_display_name' => (string) $a->author_display_name,
111 'author_first_name' => (string) $a->author_first_name,
112 'author_last_name' => (string) $a->author_last_name,
113 ];
114 }
115
116 // Grab cats, tags and terms.
117 foreach ( $xml->xpath( '/rss/channel/wp:category' ) as $term_arr ) {
118 $t = $term_arr->children( $namespaces['wp'] );
119 $category = [
120 'term_id' => (int) $t->term_id,
121 'category_nicename' => (string) $t->category_nicename,
122 'category_parent' => (string) $t->category_parent,
123 'cat_name' => (string) $t->cat_name,
124 'category_description' => (string) $t->category_description,
125 ];
126
127 foreach ( $t->termmeta as $meta ) {
128 $category['termmeta'][] = [
129 'key' => (string) $meta->meta_key,
130 'value' => (string) $meta->meta_value,
131 ];
132 }
133
134 $categories[] = $category;
135 }
136
137 foreach ( $xml->xpath( '/rss/channel/wp:tag' ) as $term_arr ) {
138 $t = $term_arr->children( $namespaces['wp'] );
139 $tag = [
140 'term_id' => (int) $t->term_id,
141 'tag_slug' => (string) $t->tag_slug,
142 'tag_name' => (string) $t->tag_name,
143 'tag_description' => (string) $t->tag_description,
144 ];
145
146 foreach ( $t->termmeta as $meta ) {
147 $tag['termmeta'][] = [
148 'key' => (string) $meta->meta_key,
149 'value' => (string) $meta->meta_value,
150 ];
151 }
152
153 $tags[] = $tag;
154 }
155
156 foreach ( $xml->xpath( '/rss/channel/wp:term' ) as $term_arr ) {
157 $t = $term_arr->children( $namespaces['wp'] );
158 $term = [
159 'term_id' => (int) $t->term_id,
160 'term_taxonomy' => (string) $t->term_taxonomy,
161 'slug' => (string) $t->term_slug,
162 'term_parent' => (string) $t->term_parent,
163 'term_name' => (string) $t->term_name,
164 'term_description' => (string) $t->term_description,
165 ];
166
167 foreach ( $t->termmeta as $meta ) {
168 $term['termmeta'][] = [
169 'key' => (string) $meta->meta_key,
170 'value' => (string) $meta->meta_value,
171 ];
172 }
173
174 $terms[] = $term;
175 }
176
177 // Grab posts.
178 foreach ( $xml->channel->item as $item ) {
179 $post = [
180 'post_title' => (string) $item->title,
181 'guid' => (string) $item->guid,
182 ];
183
184 $dc = $item->children( 'http://purl.org/dc/elements/1.1/' );
185 $post['post_author'] = (string) $dc->creator;
186
187 $content = $item->children( 'http://purl.org/rss/1.0/modules/content/' );
188 $excerpt = $item->children( $namespaces['excerpt'] );
189 $post['post_content'] = (string) $content->encoded;
190 $post['post_excerpt'] = (string) $excerpt->encoded;
191
192 $wp = $item->children( $namespaces['wp'] );
193 $post['post_id'] = (int) $wp->post_id;
194 $post['post_date'] = (string) $wp->post_date;
195 $post['post_date_gmt'] = (string) $wp->post_date_gmt;
196 $post['comment_status'] = (string) $wp->comment_status;
197 $post['ping_status'] = (string) $wp->ping_status;
198 $post['post_name'] = (string) $wp->post_name;
199 $post['status'] = (string) $wp->status;
200 $post['post_parent'] = (int) $wp->post_parent;
201 $post['menu_order'] = (int) $wp->menu_order;
202 $post['post_type'] = (string) $wp->post_type;
203 $post['post_password'] = (string) $wp->post_password;
204 $post['is_sticky'] = (int) $wp->is_sticky;
205
206 if ( isset( $wp->attachment_url ) ) {
207 $post['attachment_url'] = (string) $wp->attachment_url;
208 }
209
210 foreach ( $item->category as $c ) {
211 $att = $c->attributes();
212 if ( isset( $att['nicename'] ) ) {
213 $post['terms'][] = [
214 'name' => (string) $c,
215 'slug' => (string) $att['nicename'],
216 'domain' => (string) $att['domain'],
217 ];
218 }
219 }
220
221 foreach ( $wp->postmeta as $meta ) {
222 $post['postmeta'][] = [
223 'key' => (string) $meta->meta_key,
224 'value' => (string) $meta->meta_value,
225 ];
226 }
227
228 foreach ( $wp->comment as $comment ) {
229 $meta = [];
230 if ( isset( $comment->commentmeta ) ) {
231 foreach ( $comment->commentmeta as $m ) {
232 $meta[] = [
233 'key' => (string) $m->meta_key,
234 'value' => (string) $m->meta_value,
235 ];
236 }
237 }
238
239 $post['comments'][] = [
240 'comment_id' => (int) $comment->comment_id,
241 'comment_author' => (string) $comment->comment_author,
242 'comment_author_email' => (string) $comment->comment_author_email,
243 'comment_author_IP' => (string) $comment->comment_author_IP,
244 'comment_author_url' => (string) $comment->comment_author_url,
245 'comment_date' => (string) $comment->comment_date,
246 'comment_date_gmt' => (string) $comment->comment_date_gmt,
247 'comment_content' => (string) $comment->comment_content,
248 'comment_approved' => (string) $comment->comment_approved,
249 'comment_type' => (string) $comment->comment_type,
250 'comment_parent' => (string) $comment->comment_parent,
251 'comment_user_id' => (int) $comment->comment_user_id,
252 'commentmeta' => $meta,
253 ];
254 }
255
256 $posts[] = $post;
257 }
258
259 return [
260 'authors' => $authors,
261 'posts' => $posts,
262 'categories' => $categories,
263 'tags' => $tags,
264 'terms' => $terms,
265 'base_url' => $base_url,
266 'base_blog_url' => $base_blog_url,
267 'page_on_front' => $page_on_front,
268 'version' => $wxr_version,
269 ];
270 }
271
272 /**
273 * @param $file
274 * @param mixed ...$args
275 * @return false|string
276 */
277 public function file_get_contents( $file, ...$args ) {
278 if ( ! is_file( $file ) || ! is_readable( $file ) ) {
279 return false;
280 }
281
282 return file_get_contents( $file, ...$args );
283 }
284 }
285