.md`
* serves, and what AI crawlers index.
* - `copy` — `# Title`, blockquoted description, `Source:` line, body.
* What the "Copy page" button puts on the clipboard; no YAML,
* because a human is about to paste this into a chat box.
*
* The body is produced by running the real `the_content` filter (so blocks,
* shortcodes and embeds resolve exactly as they do on the page) and then walking
* the resulting DOM. A node-removal pass first strips chrome that is meaningless
* outside a browser — code-snippet toolbars, heading anchor links, glossary
* tooltips, scripts.
*
* @since 4.8.0
*/
class MarkdownRenderer {
/**
* Cached body Markdown (no front matter — that is assembled per context at
* request time, so one blob serves both profiles).
*/
const META_BODY = '_betterdocs_markdown';
/** Signature the cached body was generated from. */
const META_SIG = '_betterdocs_markdown_sig';
/** Cache namespace for Database::get_cache_version(). */
const CACHE_NS = 'markdown';
/**
* @var Settings
*/
protected $settings;
/**
* @var Database
*/
protected $database;
/**
* Re-entrancy guard. `the_content` can be filtered from inside a page that is
* already rendering `the_content` (a shortcode placed inside a doc), which
* would recurse forever.
*
* @var bool
*/
protected $rendering = false;
public function __construct( Settings $settings, Database $database ) {
$this->settings = $settings;
$this->database = $database;
// Bust one doc's cache when it is saved. `wp_after_insert_post` — NOT
// `save_post_docs` — because terms are not written yet at `save_post`, and
// the front matter carries the category and tags.
add_action( 'wp_after_insert_post', [ $this, 'flush_post' ], 10, 2 );
// Term renames and settings changes affect every doc's front matter, so bump
// the namespace version instead of walking the post table.
add_action( 'edited_doc_category', [ $this, 'flush_all' ] );
add_action( 'edited_doc_tag', [ $this, 'flush_all' ] );
add_action( 'delete_term', [ $this, 'flush_all' ] );
add_action( 'update_option_betterdocs_settings', [ $this, 'flush_all' ] );
}
/**
* Whether Markdown may be produced for this doc at all.
*
* The single gate both the `.md` endpoint and the AI Actions UI consult, so a
* doc that cannot be served also never renders a button pointing at it.
*
* @param \WP_Post|int|null $post
* @return bool
*/
public function can_read( $post ) {
$post = get_post( $post );
if ( ! $post instanceof \WP_Post || 'docs' !== $post->post_type ) {
return false;
}
// Drafts, pending and private docs are readable only by someone who could
// read them in the admin.
if ( 'publish' !== $post->post_status && ! current_user_can( 'read_post', $post->ID ) ) {
return false;
}
// Deliberately no `?password=` query-arg escape hatch (unlike REST\Docs):
// the cookie is already present on a front-end request, and a password in a
// GET would leak into referrers, server logs and the LLM prompt.
if ( post_password_required( $post ) ) {
return false;
}
/**
* Gate Markdown output for a doc.
*
* BetterDocs Pro hooks its Content Restriction / Access Control check here so
* a restricted doc cannot be exfiltrated through `.md`.
*
* @since 4.8.0
*
* @param bool $can_read
* @param \WP_Post $post
*/
return (bool) apply_filters( 'betterdocs_markdown_can_read', true, $post );
}
/**
* Full Markdown document for a doc.
*
* @param \WP_Post|int|null $post
* @param string $context `endpoint` | `copy`
* @return string Empty string when the caller is not entitled to the content.
*/
public function render( $post, $context = 'endpoint' ) {
$post = get_post( $post );
if ( ! $this->can_read( $post ) ) {
return '';
}
$body = $this->body( $post );
return $this->front_matter( $post, $context, $body ) . $body . "\n";
}
/**
* Body Markdown only, cached.
*
* @param \WP_Post $post
* @return string
*/
public function body( $post ) {
$signature = $this->signature( $post );
$cacheable = $this->cacheable( $post );
if ( $cacheable ) {
$cached = get_post_meta( $post->ID, self::META_BODY, true );
if ( is_string( $cached ) && '' !== $cached
&& $signature === get_post_meta( $post->ID, self::META_SIG, true ) ) {
return $cached;
}
}
$body = $this->html_to_markdown( $this->content_html( $post ) );
if ( $cacheable && '' !== $body ) {
update_post_meta( $post->ID, self::META_BODY, $body );
update_post_meta( $post->ID, self::META_SIG, $signature );
}
return $body;
}
/**
* Only published, unprotected, non-preview docs are worth persisting. Anything
* else is either transient or user-specific.
*
* The cache is shared by every reader, so it may only hold what an anonymous
* visitor would get. A logged-in reader's render can differ — membership
* plugins, "logged-in only" blocks and per-user shortcodes run inside
* the_content — so it is neither stored (it would be served to guests) nor
* read (a guest's render would be served to the member).
*
* @param \WP_Post $post
* @return bool
*/
protected function cacheable( $post ) {
$cacheable = 'publish' === $post->post_status
&& ! post_password_required( $post )
&& ! is_preview()
&& ! is_user_logged_in();
/**
* Whether a doc's rendered Markdown may be stored and reused for other
* readers. Return false for content that varies by visitor.
*
* @since 4.8.0
*
* @param bool $cacheable
* @param \WP_Post $post
*/
return (bool) apply_filters( 'betterdocs_markdown_cacheable', $cacheable, $post );
}
/**
* @param \WP_Post $post
* @return string
*/
protected function signature( $post ) {
return md5(
implode(
'|',
[
$post->post_content,
$post->post_title,
$post->post_modified_gmt,
defined( 'BETTERDOCS_VERSION' ) ? BETTERDOCS_VERSION : '',
get_locale(),
(string) $this->database->get_cache_version( self::CACHE_NS )
]
)
);
}
/**
* Drop one doc's cached Markdown.
*
* @param int $post_id
* @param \WP_Post $post
*/
public function flush_post( $post_id, $post = null ) {
if ( $post instanceof \WP_Post && 'docs' !== $post->post_type ) {
return;
}
delete_post_meta( $post_id, self::META_BODY );
delete_post_meta( $post_id, self::META_SIG );
}
/**
* Invalidate every doc's cached Markdown by bumping the namespace version.
*/
public function flush_all() {
$this->database->bump_cache_version( self::CACHE_NS );
}
/**
* Purge every stored Markdown blob. Called when the endpoint is switched off so
* a disabled feature stops occupying postmeta.
*
* @global \wpdb $wpdb
*/
public function purge() {
global $wpdb;
$wpdb->delete( $wpdb->postmeta, [ 'meta_key' => self::META_BODY ] ); // phpcs:ignore WordPress.DB.SlowDBQuery.slow_db_query_meta_key
$wpdb->delete( $wpdb->postmeta, [ 'meta_key' => self::META_SIG ] ); // phpcs:ignore WordPress.DB.SlowDBQuery.slow_db_query_meta_key
}
/* ---------------------------------------------------------------------------
* Front matter
* ------------------------------------------------------------------------- */
/**
* @param \WP_Post $post
* @param string $context
* @param string $body Already-rendered body, used to derive a description.
* @return string
*/
protected function front_matter( $post, $context, $body ) {
// get_the_title() runs the `the_title` filters, so an ampersand comes back as
// `&`. Markdown is not HTML — entities have to be decoded or they show up
// literally in the model's context.
$title = $this->decode( get_the_title( $post ) );
$permalink = get_permalink( $post );
$description = $this->decode( $this->description( $post, $body ) );
$data = [
'title' => $title,
'description' => $description,
'url' => $permalink,
'updated' => get_post_modified_time( 'Y-m-d', true, $post ),
'category' => $this->category_path( $post ),
'tags' => wp_get_post_terms( $post->ID, 'doc_tag', [ 'fields' => 'names' ] )
];
if ( is_wp_error( $data['tags'] ) ) {
$data['tags'] = [];
}
$data['tags'] = array_map( [ $this, 'decode' ], $data['tags'] );
/**
* Filter the Markdown front-matter data before it is serialised.
*
* @since 4.8.0
*
* @param array $data
* @param \WP_Post $post
* @param string $context `endpoint` | `copy`
*/
$data = (array) apply_filters( 'betterdocs_markdown_front_matter', $data, $post, $context );
if ( 'copy' === $context ) {
// Human-facing: no YAML, because this is about to be pasted into a chat.
$out = '# ' . $data['title'] . "\n\n";
if ( ! empty( $data['description'] ) ) {
$out .= '> ' . $data['description'] . "\n\n";
}
if ( ! empty( $data['url'] ) ) {
$out .= 'Source: ' . $data['url'] . "\n\n";
}
return $out;
}
$out = "---\n";
foreach ( $data as $key => $value ) {
if ( is_array( $value ) ) {
if ( empty( $value ) ) {
continue;
}
$out .= $key . ":\n";
foreach ( $value as $item ) {
$out .= ' - ' . $this->yaml_scalar( $item ) . "\n";
}
continue;
}
if ( '' === (string) $value ) {
continue;
}
$out .= $key . ': ' . $this->yaml_scalar( $value ) . "\n";
}
$out .= "---\n\n";
$out .= '# ' . $data['title'] . "\n\n";
return $out;
}
/**
* The doc's own excerpt when it has one, otherwise the opening of the body.
*
* `get_the_excerpt()` is avoided on purpose — it fires the `the_excerpt`
* filters, which other plugins use to append read-more markup.
*
* @param \WP_Post $post
* @param string $body
* @return string
*/
protected function description( $post, $body ) {
if ( ! empty( $post->post_excerpt ) ) {
return trim( wp_strip_all_tags( $post->post_excerpt ) );
}
return trim( wp_trim_words( wp_strip_all_tags( $body ), 40, '…' ) );
}
/**
* Full hierarchical category path, e.g. "Getting Started / Installation".
*
* @param \WP_Post $post
* @return string
*/
protected function category_path( $post ) {
$terms = get_the_terms( $post->ID, 'doc_category' );
if ( is_wp_error( $terms ) || empty( $terms ) ) {
return '';
}
$term = $terms[0];
$names = [ $term->name ];
foreach ( get_ancestors( $term->term_id, 'doc_category', 'taxonomy' ) as $ancestor_id ) {
$ancestor = get_term( $ancestor_id, 'doc_category' );
if ( $ancestor && ! is_wp_error( $ancestor ) ) {
array_unshift( $names, $ancestor->name );
}
}
return $this->decode( implode( ' / ', $names ) );
}
/**
* Turn HTML entities back into the characters they stand for.
*
* @param string $text
* @return string
*/
protected function decode( $text ) {
return html_entity_decode( (string) $text, ENT_QUOTES | ENT_HTML5, 'UTF-8' );
}
/**
* Quote a YAML scalar only when it would otherwise break the mapping.
*
* @param mixed $value
* @return string
*/
protected function yaml_scalar( $value ) {
$value = (string) $value;
if ( '' === $value ) {
return '""';
}
// Line breaks would end the scalar and start a new key; inside double
// quotes a backslash starts an escape, so it is escaped first.
if ( preg_match( '/[:#\-\[\]\{\}&\*!\|>\'"%@`\r\n\t]/', $value ) || preg_match( '/^\s|\s$/', $value ) ) {
return '"' . str_replace( [ '\\', '"', "\r", "\n", "\t" ], [ '\\\\', '\"', '\r', '\n', '\t' ], $value ) . '"';
}
return $value;
}
/* ---------------------------------------------------------------------------
* the_content pipeline
* ------------------------------------------------------------------------- */
/**
* Rendered HTML for a doc, with the filters that only make sense in a browser
* temporarily unhooked.
*
* @param \WP_Post $post
* @return string
*/
protected function content_html( $post ) {
if ( $this->rendering ) {
return '';
}
$this->rendering = true;
global $wp_query;
$prev_post = isset( $GLOBALS['post'] ) ? $GLOBALS['post'] : null;
$prev_inloop = isset( $wp_query->in_the_loop ) ? $wp_query->in_the_loop : false;
// Dynamic blocks and Pro's glossary wrapper read the global $post, so the
// loop has to be standing before `the_content` runs.
$GLOBALS['post'] = $post;
if ( isset( $wp_query ) ) {
$wp_query->in_the_loop = true;
}
setup_postdata( $post );
// Presentation-only filters. `do_blocks`, `wpautop`, `do_shortcode` and
// `WP_Embed::autoembed` all stay — removing autoembed makes do_shortcode
// delete the bare URL outright instead of leaving it as text.
$removed = [
[ 'wptexturize', 10 ], // smart quotes corrupt CLI and code samples
[ 'capital_P_dangit', 11 ], // rewrites "Wordpress" inside code samples
[ 'wp_filter_content_tags', 12 ], // srcset/sizes/loading/decoding noise
[ 'convert_smilies', 20 ] // ":)" becomes
];
foreach ( $removed as $filter ) {
remove_filter( 'the_content', $filter[0], $filter[1] );
}
try {
$html = apply_filters( 'the_content', $post->post_content );
} finally {
foreach ( $removed as $filter ) {
add_filter( 'the_content', $filter[0], $filter[1] );
}
if ( isset( $wp_query ) ) {
$wp_query->in_the_loop = $prev_inloop;
}
$GLOBALS['post'] = $prev_post;
wp_reset_postdata();
$this->rendering = false;
}
return (string) $html;
}
/* ---------------------------------------------------------------------------
* HTML -> Markdown
* ------------------------------------------------------------------------- */
/**
* Convert an HTML fragment to Markdown by walking the DOM.
*
* @param string $html
* @return string
*/
public function html_to_markdown( $html ) {
if ( '' === trim( (string) $html ) ) {
return '';
}
// ext-dom is not a declared requirement (see composer.json), and
// Core\SampleDocBuilder already guards for it. Degrade to plain text rather
// than fataling on a host without php-xml.
if ( ! class_exists( '\DOMDocument' ) ) {
return trim( wp_strip_all_tags( $html ) );
}
$dom = new \DOMDocument();
libxml_use_internal_errors( true );
$dom->loadHTML(
'
/, where Markdown is not escaped.
* @return string
*/
protected function children_md( $node, $in_code = false ) {
$out = '';
foreach ( $node->childNodes as $child ) {
$out .= $this->node_to_md( $child, $in_code );
}
return $out;
}
/**
* @param \DOMNode $node
* @param bool $in_code
* @return string
*/
protected function node_to_md( $node, $in_code = false ) {
if ( XML_TEXT_NODE === $node->nodeType ) {
$text = preg_replace( '/\s+/', ' ', $node->nodeValue );
return $in_code ? $text : $this->escape( $text );
}
if ( XML_ELEMENT_NODE !== $node->nodeType ) {
return '';
}
$tag = strtolower( $node->nodeName );
// BetterDocs code snippets carry their own toolbar and line-number column;
// the toolbar is already pruned, so emit just the code with its language.
if ( 'div' === $tag && $this->node_has_class( $node, 'betterdocs-code-snippet-wrapper' ) ) {
return $this->fence( $node->textContent, $node->getAttribute( 'data-language' ) );
}
$inner = $this->children_md( $node, $in_code );
switch ( $tag ) {
case 'h1':
case 'h2':
case 'h3':
case 'h4':
case 'h5':
case 'h6':
return "\n\n" . str_repeat( '#', (int) substr( $tag, 1 ) ) . ' ' . trim( $inner ) . "\n\n";
case 'p':
return "\n\n" . trim( $inner ) . "\n\n";
case 'br':
return " \n";
case 'hr':
return "\n\n---\n\n";
case 'strong':
case 'b':
return '' === trim( $inner ) ? '' : '**' . trim( $inner ) . '**';
case 'em':
case 'i':
return '' === trim( $inner ) ? '' : '*' . trim( $inner ) . '*';
case 'del':
case 's':
case 'strike':
return '' === trim( $inner ) ? '' : '~~' . trim( $inner ) . '~~';
case 'mark':
return trim( $inner );
case 'kbd':
case 'samp':
return '`' . trim( $node->textContent ) . '`';
case 'sup':
return '^' . trim( $inner );
case 'sub':
return '~' . trim( $inner );
case 'a':
$href = $node->getAttribute( 'href' );
$text = trim( $inner );
if ( '' === $href ) {
return $text;
}
return '[' . ( '' === $text ? $href : $text ) . '](' . $href . ')';
case 'img':
$src = $node->getAttribute( 'src' );
return '' === $src ? '' : '';
case 'figure':
return "\n\n" . trim( $inner ) . "\n\n";
case 'figcaption':
return '' === trim( $inner ) ? '' : "\n" . trim( $inner ) . "\n";
case 'iframe':
$src = $node->getAttribute( 'src' );
if ( '' === $src ) {
return '';
}
$label = $node->getAttribute( 'title' );
return "\n\n[" . ( '' === $label ? $src : $this->escape( $label ) ) . '](' . $src . ")\n\n";
case 'code':
// Inline code only; fenced blocks are handled by .
return '`' . trim( $node->textContent ) . '`';
case 'pre':
return $this->fence( $node->textContent, $this->language_of( $node ) );
case 'blockquote':
$quote = trim( $this->children_md( $node, $in_code ) );
return "\n\n" . preg_replace( '/^/m', '> ', $quote ) . "\n\n";
case 'ul':
return "\n\n" . $this->list_md( $node, false ) . "\n\n";
case 'ol':
return "\n\n" . $this->list_md( $node, true ) . "\n\n";
case 'table':
return $this->table_md( $node );
case 'dl':
return "\n\n" . trim( $inner ) . "\n\n";
case 'dt':
return "\n" . '**' . trim( $inner ) . '**' . "\n";
case 'dd':
return ': ' . trim( $inner ) . "\n";
default:
return $inner; // unwrap unknown containers (div/span/section/…)
}
}
/**
* Escape the Markdown metacharacters that would otherwise reformat prose.
*
* Intra-word underscores are deliberately left alone: CommonMark does not treat
* them as emphasis, so escaping them only makes identifiers like `my_var` uglier
* for the model reading them.
*
* @param string $text
* @return string
*/
protected function escape( $text ) {
return preg_replace( '/([\\\\`\*\[\]])/', '\\\\$1', (string) $text );
}
/**
* @param \DOMNode $node
* @param string $class
* @return bool
*/
protected function node_has_class( $node, $class ) {
if ( ! $node instanceof \DOMElement ) {
return false;
}
return in_array( $class, preg_split( '/\s+/', trim( $node->getAttribute( 'class' ) ) ), true );
}
/**
* Language hint for a fenced block, read from the usual `language-*` class on
* the or its child .
*
* @param \DOMNode $node
* @return string
*/
protected function language_of( $node ) {
$candidates = [ $node ];
foreach ( $node->childNodes as $child ) {
if ( XML_ELEMENT_NODE === $child->nodeType && 'code' === strtolower( $child->nodeName ) ) {
$candidates[] = $child;
}
}
foreach ( $candidates as $candidate ) {
if ( ! $candidate instanceof \DOMElement ) {
continue;
}
if ( preg_match( '/(?:language|lang|brush:)[-\s]([a-z0-9#+_-]+)/i', $candidate->getAttribute( 'class' ), $m ) ) {
return strtolower( $m[1] );
}
}
return '';
}
/**
* @param string $code
* @param string $language
* @return string
*/
protected function fence( $code, $language = '' ) {
$code = rtrim( ltrim( (string) $code, "\r\n" ) );
if ( '' === trim( $code ) ) {
return '';
}
// Use a longer fence when the snippet itself contains a triple backtick.
$fence = preg_match( '/^\s*```/m', $code ) ? '````' : '```';
return "\n\n" . $fence . $language . "\n" . $code . "\n" . $fence . "\n\n";
}
/**
* @param \DOMNode $node The / element.
* @param bool $ordered
* @return string
*/
protected function list_md( $node, $ordered ) {
$lines = [];
$index = 1;
foreach ( $node->childNodes as $li ) {
if ( XML_ELEMENT_NODE !== $li->nodeType || 'li' !== strtolower( $li->nodeName ) ) {
continue;
}
$marker = $ordered ? ( $index++ . '. ' ) : '- ';
$content = trim( $this->children_md( $li ) );
// Collapse blank lines inside the item so a nested list attaches directly
// under its parent marker, then indent continuation lines to the marker
// width (2 for "- ", 3 for "1. ").
$content = preg_replace( "/\n{2,}/", "\n", $content );
$content = str_replace( "\n", "\n" . str_repeat( ' ', strlen( $marker ) ), $content );
$lines[] = $marker . $content;
}
return implode( "\n", $lines );
}
/**
* Convert a to a GFM pipe table. Falls back to unwrapped text when the
* table has no rows we can line up.
*
* @param \DOMNode $node
* @return string
*/
protected function table_md( $node ) {
$dom = $node->ownerDocument;
$rows = [];
$tr_nodes = ( new \DOMXPath( $dom ) )->query( './/tr', $node );
if ( ! $tr_nodes || 0 === $tr_nodes->length ) {
return "\n\n" . trim( $this->children_md( $node ) ) . "\n\n";
}
foreach ( $tr_nodes as $tr ) {
$cells = [];
foreach ( $tr->childNodes as $cell ) {
if ( XML_ELEMENT_NODE !== $cell->nodeType ) {
continue;
}
$name = strtolower( $cell->nodeName );
if ( 'td' !== $name && 'th' !== $name ) {
continue;
}
// A pipe table cell is a single line; a literal pipe must be escaped.
$text = trim( preg_replace( "/\s*\n\s*/", ' ', $this->children_md( $cell ) ) );
$cells[] = str_replace( '|', '\|', $text );
}
if ( $cells ) {
$rows[] = $cells;
}
}
if ( ! $rows ) {
return '';
}
$columns = max( array_map( 'count', $rows ) );
$header = array_shift( $rows );
$header = array_pad( $header, $columns, '' );
$out = '| ' . implode( ' | ', $header ) . " |\n";
$out .= '| ' . implode( ' | ', array_fill( 0, $columns, '---' ) ) . " |\n";
foreach ( $rows as $row ) {
$out .= '| ' . implode( ' | ', array_pad( $row, $columns, '' ) ) . " |\n";
}
return "\n\n" . rtrim( $out ) . "\n\n";
}
}