Invision Power Services, Inc.
* @copyright (c) Invision Power Services, Inc.
* @license https://www.invisioncommunity.com/legal/standards/
* @package Invision Community
* @since 12 Jun 2013
*/
namespace IPS\Text;
/* To prevent PHP errors (extending class does not exist) revealing path */
use DOMElement;
use DOMNode;
use DOMText;
use DOMXpath;
use Exception;
use HTMLPurifier;
use IPS\Application;
use IPS\core\Profanity;
use IPS\Db;
use IPS\forums\Topic\Post;
use IPS\Http\Url;
use IPS\IPS;
use IPS\Member;
use IPS\Settings;
use IPS\Xml\DOMDocument;
use OutOfRangeException;
use function count;
use function defined;
use function function_exists;
use function in_array;
use function is_array;
use function is_string;
use function substr;
use const IPS\DEFAULT_REQUEST_TIMEOUT;
use const IPS\ROOT_PATH;
if ( !defined( '\IPS\SUITE_UNIQUE_KEY' ) )
{
header( ( $_SERVER['SERVER_PROTOCOL'] ?? 'HTTP/1.0' ) . ' 403 Forbidden' );
exit;
}
/**
* Text Parser
*/
class ConverterParser extends Parser
{
/**
* @brief Regex for detecting email addresses
*/
const EMAIL_REGEX = '[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,9}';
/* !Parser: Bootstrap */
/**
* @brief If parsing BBCode, the supported BBCode tags
*/
public ?array $bbcodes = NULL;
/**
* Constructor
*
* @param bool $bbcode Parse BBCode?
* @param mixed $attachIds array of ID numbers to idenfity content for attachments if the content has been saved - the first two must be int or null, the third must be string or null. If content has not been saved yet, an MD5 hash used to claim attachments after saving.
* @param Member|null $member The member posting, NULL will use currently logged in member.
* @param bool|string $area If parsing BBCode or attachments, the Editor area we're parsing in. e.g. "core_Signatures". A boolean value will allow or disallow all BBCodes that are dependant on area.
* @param bool $filterProfanity Remove profanity?
* @param bool $cleanHtml If TRUE, HTML will be cleaned through HTMLPurifier
* @param callback|null $htmlPurifierConfig A function which will be passed the HTMLPurifier_Config object to customise it - see example
* @param bool $parseAcronyms Parse acronyms?
* @param ?int $attachIdsLang Language ID number if this Editor is part of a Translatable field.
* @return void
*/
public function __construct(bool $bbcode=FALSE, mixed $attachIds=NULL, Member $member=NULL, bool|string $area=FALSE, bool $filterProfanity=TRUE, bool $cleanHtml=TRUE, callable $htmlPurifierConfig=NULL, bool $parseAcronyms=TRUE, int $attachIdsLang=NULL )
{
/* Set the Member */
$this->member = $member ?: Member::loggedIn();
/* Set the member and area */
if ( $bbcode or $attachIds )
{
$this->area = $area;
}
/* Get available BBCodes */
if ( $bbcode )
{
$this->bbcodes = static::bbcodeTags( $this->member, $this->area );
}
/* Get attachments */
$this->attachIds = $attachIds;
$this->attachIdsLang = $attachIdsLang;
if( $attachIds !== NULL )
{
$where = array( array( 'location_key=?', $area ) );
if ( is_array( $attachIds ) )
{
$i = 1;
foreach ( $attachIds as $id )
{
$where[] = array("id{$i}=?", $id);
$i++;
}
}
elseif ( is_string( $attachIds ) )
{
$where[] = array( 'temp=?', $attachIds );
}
$this->existingAttachments = iterator_to_array( Db::i()->select( '*', 'core_attachments_map', $where )->setKeyField( 'attachment_id' ) );
$this->mappedAttachments = array_keys( $this->existingAttachments );
}
/* Get profanity filters */
if ( $filterProfanity )
{
foreach( Profanity::getProfanity() AS $profanity )
{
if ( $profanity->action == 'swap' )
{
if ( $profanity->m_exact )
{
$this->exactProfanity[ $profanity->type ] = $profanity->swop;
}
else
{
$this->looseProfanity[ $profanity->type ] = $profanity->swop;
}
}
}
}
/* Get HTMLPurifier Configuration */
if ( $cleanHtml )
{
if ( !function_exists('idn_to_ascii') )
{
IPS::$PSR0Namespaces['TrueBV'] = ROOT_PATH . "/system/3rd_party/php-punycode";
require_once ROOT_PATH . "/system/3rd_party/php-punycode/polyfill.php";
}
require_once ROOT_PATH . "/system/3rd_party/HTMLPurifier/HTMLPurifier.auto.php";
$this->htmlPurifier = new HTMLPurifier( $this->_htmlPurifierConfiguration( $htmlPurifierConfig ) );
}
/* Get acronyms */
if ( $parseAcronyms )
{
$this->caseSensitiveAcronyms = iterator_to_array( Db::i()->select( array( 'a_short', 'a_long', 'a_type' ), 'core_acronyms', array( 'a_casesensitive=1' ) )->setKeyField( 'a_short' ) );
$this->caseInsensitiveAcronyms = array();
foreach ( Db::i()->select( array( 'a_short', 'a_long', 'a_type' ), 'core_acronyms', array( 'a_casesensitive=0' ) )->setKeyField( 'a_short' ) as $k => $v )
{
$this->caseInsensitiveAcronyms[ mb_strtolower( $k ) ] = $v;
}
}
}
/**
* Force bbcode parsing enabled
*
* @return void
*/
public function __set( $name, $enabled = TRUE )
{
if( $name == 'forceBbcodeEnabled' )
{
$this->forceBbcodeEnabled = $enabled;
if( $enabled )
{
$this->bbcodes = static::bbcodeTags( $this->member, $this->area );
}
else
{
$this->bbcodes = NULL;
}
}
}
/**
* @brief The closing BBCode tags we are looking for and how many are open
*/
protected array $closeTagsForOpenBBCode = array();
/**
* @brief Open Inline BBCode tags
*/
protected array $openInlineBBCode = array();
/**
* @brief All open Block-Level BBCode tags
*/
protected array $openBlockBBCodeByTag = array();
/**
* @brief All open Block-Level BBCode tags in the order they were created
*/
protected array $openBlockBBCodeInOrder = array();
/**
* @brief Open Block-Level BBCode tags
*/
protected ?int $openBlockDepth = NULL;
/**
* @brief Force BBCode enabled (used by 4.0 upgrader parsing and converters)
* @see set_forceBbcodeEnabled()
*/
protected bool $forceBbcodeEnabled = FALSE;
/**
* @brief This is used to stop BBCode parsing temporarily (such as in [code] tags)
*/
protected bool $bbcodeParse = TRUE;
/**
* @brief If we have opened a BBCode tag which we don't parse other BBCode inside, the string at which we will resume parsing
*/
protected ?string $resumeBBCodeParsingOn = NULL;
/**
* Parse BBCode, Profanity, etc. by loading into a DOMDocument
*
* @param string $value HTML to parse
* @return string
*/
protected function _parseContent( string $value ): string
{
/* This fix resolves an issue using
mode where BBCode tags are wrapped in P tags like so:
[tag]
Content
[/tag]
tags. The fix just removes thetags inside block BBCode tags, so our example ends up parsing like so:
[tag]
Content
[/tag]
tags are converted to
to prevent parser confusion */
preg_match_all( '#\[(' . implode( '|', $blockTags ) . ')\](.+?)\[/\1\]#si', $value, $matches, PREG_SET_ORDER );
foreach( $matches as $id => $match )
{
$value = str_replace( $match[0], preg_replace( '#
]+?)?'.'>#i', '
', $match[0] ), $value );
}
}
}
/* The editor button just drops in a
Page one
Page two
Page one
Page two
*/ /* Parse */ $parser = new DOMParser( array( $this, '_parseDomElement' ), array( $this, '_parseDomText' ) ); $document = $parser->parseValueIntoDocument( $value ); /* [page] tags need to be handled specially */ if ( $this->bbcodes !== NULL and $this->containsPageTags ) { $body = DOMParser::getDocumentBody( $document ); $bodyWithPages = $this->_parseContentWithSeparationTag( $body, function ( \DOMDocument $document ) { $mainDiv = $document->createElement('div'); $mainDiv->setAttribute( 'data-controller', 'core.front.core.articlePages' ); return $mainDiv; }, function ( \DOMDocument $document ) { $subDiv = $document->createElement('div'); $subDiv->setAttribute( 'data-role', 'contentPage' ); $hr = $document->createElement('hr'); $hr->setAttribute( 'data-role', 'contentPageBreak' ); $subDiv->appendChild( $hr ); return $subDiv; }, '[page]' ); $newBody = new DOMElement('body'); $body->parentNode->replaceChild( $newBody, $body ); $newBody->appendChild( $bodyWithPages ); } /* Return */ return DOMParser::getDocumentBodyContents( $document ); } /** * Parse HTML element (e.g. ,\t" so also strip any whitespace after a break CKEditor doesn't actually have a way to indent individual lines */ if ( ( $previousSibling = $textNode->previousSibling and $previousSibling instanceof DOMElement and $previousSibling->tagName == 'br' ) or ( $textNode->parentNode instanceof DOMElement and $textNode->parentNode->tagName == 'p' and $textNode->parentNode->lastChild and $textNode->parentNode->lastChild instanceof DOMElement and $textNode->parentNode->lastChild->tagName == 'br' ) ) { $text = preg_replace( '/^\s/', '', $text ); } /* Insert */ $parent->appendChild( new DOMText( $text ) ); } ) ), 21, -6 ); /* Create a new
with those contents */
$return = $originalNode->ownerDocument->createElement( 'pre' );
$return->appendChild( new DOMText( html_entity_decode( $contents ) ) ); // We have to decode HTML entities otherwise they'll be double-encoded. Test with "[code]Test[/code]"
$return->setAttribute( 'class', 'ipsCode' );
return $return;
} );
$return['code'] = $code;
$return['codebox'] = $code;
$return['html'] = $code;
$return['php'] = $code;
$return['sql'] = $code;
$return['xml'] = $code;
/* Color */
$return['color'] = array( 'tag' => 'span', 'attributes' => array( 'style' => 'color:{option}' ), 'allowOption' => TRUE );
/* Font */
$return['font'] = array( 'tag' => 'span', 'attributes' => array( 'style' => 'font-family:{option}' ), 'allowOption' => TRUE );
/* HR */
$return['hr'] = array( 'tag' => 'hr', 'single' => TRUE, 'allowOption' => FALSE );
/* Image */
$return['img'] = array( 'tag' => 'img', 'attributes' => array( 'src' => '{option}', 'class' => 'ipsImage' ), 'single' => TRUE, 'allowOption' => TRUE );
/* Indent */
$return['indent'] = array( 'tag' => 'div', 'attributes' => array( 'style' => 'margin-left:{option}px' ), 'block' => FALSE, 'defaultOption' => 25, 'allowOption' => TRUE );
/* Italics */
$return['i'] = array( 'tag' => 'em', 'allowOption' => FALSE );
/* Justify */
$return['left'] = array( 'tag' => 'div', 'attributes' => array( 'style' => 'text-align:left' ), 'block' => TRUE, 'allowOption' => FALSE );
$return['center'] = array( 'tag' => 'div', 'attributes' => array( 'style' => 'text-align:center' ), 'block' => TRUE, 'allowOption' => FALSE );
$return['right'] = array( 'tag' => 'div', 'attributes' => array( 'style' => 'text-align:right' ), 'block' => TRUE, 'allowOption' => FALSE );
/* Links */
/* Email */
$return['email'] = array( 'tag' => 'a', 'attributes' => array( 'href' => 'mailto:{option}' ), 'allowOption' => TRUE );
/* Member */
$return['member'] = array(
'tag' => 'a',
'attributes'=> array( 'contenteditable' => 'false', 'data-ipsHover' => '' ),
'callback' => function( DOMElement $node, $matches, \DOMDocument $document )
{
try
{
$member = Member::load( $matches[2], 'name' );
if ( $member->member_id != 0 )
{
$node->setAttribute( 'href', $member->url() );
$node->setAttribute( 'data-ipsHover-target', $member->url()->setQueryString( 'do', 'hovercard' ) );
$node->setAttribute( 'data-mentionid', $member->member_id );
$node->appendChild( $document->createTextNode( '@' . $member->name ) );
}
}
catch ( Exception $e ) {}
return $node;
},
'single' => TRUE,
'allowOption' => TRUE
);
/* Links */
$return['url'] = array( 'tag' => 'a', 'attributes' => array( 'href' => '{option}' ), 'allowOption' => TRUE );
/* List */
$return['list'] = array(
'tag' => 'ul',
'callback' => function( $node, $matches, $document )
{
/* Set the main attributes for our element */ $node->setAttribute( 'class', 'ipsQuote' ); $node->setAttribute( 'data-ipsQuote', '' ); if ( isset( $options['name'] ) and $options['name'] ) { $node->setAttribute( 'data-ipsQuote-username', $options['name'] ); } if ( isset( $options['date'] ) and $options['date'] ) { $node->setAttribute( 'data-ipsQuote-timestamp', strtotime( $options['date'] ) ); } if ( Application::appIsEnabled('forums') and isset( $options['post'] ) and $options['post'] ) { try { $post = Post::load( $options['post'] ); $node->setAttribute( 'data-ipsQuote-contentapp', 'forums' ); $node->setAttribute( 'data-ipsQuote-contenttype', 'forums' ); $node->setAttribute( 'data-ipsQuote-contentclass', 'forums_Topic' ); $node->setAttribute( 'data-ipsQuote-contentid', $post->item()->tid ); $node->setAttribute( 'data-ipsQuote-contentcommentid', $post->pid ); } catch ( OutOfRangeException $e ) {} } /* Create the citation element */ $citation = $document->createElement('div'); $citation->setAttribute( 'class', 'ipsQuote_citation' ); $node->appendChild( $citation ); /* Create the content element */ $contents = $document->createElement('div'); $contents->setAttribute( 'class', 'ipsQuote_contents ipsClearfix' ); $node->appendChild( $contents ); return $node; }, 'getBlockContentElement' => function( DOMElement $node ) { foreach ( $node->childNodes as $child ) { if ( $child instanceof DOMElement and mb_strpos( $child->getAttribute('class'), 'ipsQuote_contents' ) !== FALSE ) { return $child; } } return $node; }, 'block' => TRUE, 'allowOption' => TRUE ); /* Size */ $return['size'] = array( 'tag' => 'span', 'callback' => function( $node, $matches ) { switch ( $matches[2] ) { case 1: $node->setAttribute( 'style', 'font-size:8px' ); break; case 2: $node->setAttribute( 'style', 'font-size:10px' ); break; case 3: $node->setAttribute( 'style', 'font-size:12px' ); break; case 4: $node->setAttribute( 'style', 'font-size:14px' ); break; case 5: $node->setAttribute( 'style', 'font-size:18px' ); break; case 6: $node->setAttribute( 'style', 'font-size:24px' ); break; case 7: $node->setAttribute( 'style', 'font-size:36px' ); break; case 8: $node->setAttribute( 'style', 'font-size:48px' ); break; } return $node; }, 'allowOption' => TRUE ); /* Spoiler */ $return['spoiler'] = array( 'tag' => 'div', 'callback' => function( DOMElement $node, $matches, \DOMDocument $document ) { /* Set the main attributes for ourelement */ $node->setAttribute( 'class', 'ipsSpoiler' ); $node->setAttribute( 'data-ipsSpoiler', '' ); /* Create the citation element */ $header = $document->createElement('div'); $header->setAttribute( 'class', 'ipsSpoiler_header' ); $node->appendChild( $header ); $headerSpan = $document->createElement('span'); $header->appendChild( $headerSpan ); /* Create the content element */ $contents = $document->createElement('div'); $contents->setAttribute( 'class', 'ipsSpoiler_contents' ); $node->appendChild( $contents ); return $node; }, 'getBlockContentElement' => function( DOMElement $node ) { foreach ( $node->childNodes as $child ) { if ( $child instanceof DOMElement and $child->getAttribute('class') == 'ipsSpoiler_contents' ) { return $child; } } return $node; }, 'block' => TRUE, 'allowOption' => FALSE ); /* Strike */ $return['s'] = array( 'tag' => 'span', 'attributes' => array( 'style' => 'text-decoration:line-through' ), 'allowOption' => FALSE ); $return['strike'] = array( 'tag' => 'span', 'attributes' => array( 'style' => 'text-decoration:line-through' ), 'allowOption' => FALSE ); /* Subscript */ $return['sub'] = array( 'tag' => 'sub', 'allowOption' => FALSE ); /* Superscript */ $return['sup'] = array( 'tag' => 'sup', 'allowOption' => FALSE ); /* Underline */ $return['u'] = array( 'tag' => 'span', 'attributes' => array( 'style' => 'text-decoration:underline' ), 'allowOption' => FALSE ); return $return; } /* !Utility Methods */ /** * Parse statically * * @param string $value The value to parse * @param array|null $attachIds array of ID numbers to idenfity content for attachments if the content has been saved - the first two must be int or null, the third must be string or null. If content has not been saved yet, an MD5 hash used to claim attachments after saving. * @param Member|null $member The member posting, NULL will use currently logged in member. * @param bool|string $area If parsing BBCode or attachments, the Editor area we're parsing in. e.g. "core_Signatures". A boolean value will allow or disallow all BBCodes that are dependant on area. * @param bool $filterProfanity Remove profanity? * @param callback|null $htmlPurifierConfig A function which will be passed the HTMLPurifier_Config object to customise it - see example * @param bool $bbcode Parse BBCode? * @param bool $cleanHtml If TRUE, HTML will be cleaned through HTMLPurifier * * @return string * @see __construct */ public static function parseStatic( string $value, array $attachIds=NULL, Member $member=NULL, bool|string $area=FALSE, bool $filterProfanity=TRUE, callable $htmlPurifierConfig=NULL, bool $bbcode=FALSE, bool $cleanHtml=TRUE ): string { $obj = new static($bbcode, $attachIds, $member, $area, $filterProfanity, $cleanHtml, $htmlPurifierConfig); return $obj->parse( $value ); } /** * Rebuild rel tag contents for posts * * @param string $textContent The content of the text, and they say comments aren't useful * @param Member $member The author of the content * @return mixed FALSE or changed content */ public static function rebuildUrlRels( string $textContent, Member $member ): mixed { $obj = new static(TRUE, NULL, $member); $rebuilt = FALSE; /* Create DOMDocument */ $content = new DOMDocument( '1.0', 'UTF-8' ); @$content->loadHTML( DOMDocument::wrapHtml( $textContent ) ); $xpath = new DOMXpath( $content ); foreach( $xpath->query('//a') as $link ) { if ( $link->getAttribute('href') ) { try { $url = Url::createFromString( str_replace( array( '<___base_url___>', '%7B___base_url___%7D' ), rtrim( Settings::i()->base_url, '/' ), $link->getAttribute('href') ) ); $rels = $obj->_getRelAttributes( $url ); $link->setAttribute( 'rel', implode( ' ', $rels ) ); $rebuilt = TRUE; } catch( Exception $e ) { } } } if ( $rebuilt ) { $value = $content->saveHTML(); $value = preg_replace( '/]+?)>/i', '', preg_replace( '/^/', '', str_replace( array( '', '', '', '', '', '' ), '', $value ) ) ); /* Replace file storage tags */ $value = preg_replace( '/<fileStore\.([\d\w\_]+?)>/i', '', $value ); /* DOMDocument::saveHTML will encode the base_url brackets, so we need to make sure it's in the expected format. */ return str_replace( '<___base_url___>', '<___base_url___>', $value ); } return FALSE; } }