Invision Power Services, Inc. * @copyright (c) Invision Power Services, Inc. * @license https://www.invisioncommunity.com/legal/standards/ * @package Invision Community * @since 12 Jun 2013 */ namespace IPS\Text; /* To prevent PHP errors (extending class does not exist) revealing path */ use DateInterval; use DomainException; use DOMElement; use DOMNode; use DOMText; use DOMXpath; use Exception; use HTMLPurifier; use HTMLPurifier_AttrDef_CSS_Border; use HTMLPurifier_AttrDef_CSS_Composite; use HTMLPurifier_AttrDef_CSS_Length; use HTMLPurifier_AttrDef_CSS_Multiple; use HTMLPurifier_AttrDef_CSS_Percentage; use HTMLPurifier_AttrDef_Enum; use HTMLPurifier_AttrDef_HTML_Bool; use HTMLPurifier_AttrDef_URI; use HTMLPurifier_Config; use HTMLPurifier_CSSDefinition; use HTMLPurifier_DefinitionCacheFactory; use HTMLPurifier_HTMLDefinition; use InvalidArgumentException; use IPS\Application; use IPS\Content; use IPS\core\Profanity; use IPS\Data\Cache; use IPS\DateTime; use IPS\Db; use IPS\File; use IPS\Http\Url; use IPS\Http\Url\Internal; use IPS\IPS; use IPS\Lang; use IPS\Log; use IPS\Member; use IPS\Member\Club; use IPS\Node\Model; use IPS\Patterns\ActiveRecordIterator; use IPS\Platform\Bridge; use IPS\Request; use IPS\Settings; use IPS\Theme; use IPS\Xml\DOMDocument; use OutOfRangeException; use UnderflowException; use UnexpectedValueException; use function array_key_exists; use function chr; use function count; use function defined; use function explode; use function function_exists; use function implode; use function in_array; use function intval; use function is_array; use function is_string; use function json_decode; use function md5; use function mt_rand; use function preg_quote; use function str_replace; use function substr; use const IPS\DEFAULT_REQUEST_TIMEOUT; use const IPS\ROOT_PATH; use const PHP_URL_SCHEME; if ( !defined( '\IPS\SUITE_UNIQUE_KEY' ) ) { header( ( $_SERVER['SERVER_PROTOCOL'] ?? 'HTTP/1.0' ) . ' 403 Forbidden' ); exit; } /** * Text Parser */ class Parser { /** * Get the regex pattern used to match emojis in text * * @return string */ public static function getEmojiRegex() : string { static $output = null; if ( !is_string( $output ) OR !$output ) { $cacheKey = 'IPS_TEXT_PARSER_REGEX'; try { $output = Cache::i()->getWithExpire( $cacheKey, true ); } catch ( OutOfRangeException ) {} if ( !$output ) { $output = ""; $data = json_decode( file_get_contents( ROOT_PATH . '/applications/core/data/emojiRegex.json' ), true ); if ( is_array( $data ) and isset( $data['emojiRegex'] ) and is_string( $data['emojiRegex'] ) ) { $output = $data["emojiRegex"]; } Cache::i()->storeWithExpire( $cacheKey, $output, ( new DateTime() )->add( new DateInterval( 'P1D' ) ) ); } } return $output; } /** * @brief Regex for detecting email addresses */ const EMAIL_REGEX = '[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,9}'; /* !Parser: Bootstrap */ /** * @brief Attachment IDs */ protected mixed $attachIds = NULL; /** * @brief Attachment Lang */ protected ?int $attachIdsLang = NULL; /** * @brief Rows from core_attachments_map containing attachments which belong to the content being edited - as they are found by the parser, they will be removed so we are left with attachments that have been removed * @var array (array key is attachment ID, value is the row from core_attachments_map) */ public array $existingAttachments = array(); /** * @brief Attachment IDs * @var array (array of attachment IDs that belong to the content being edited) */ public array $mappedAttachments = array(); /** * @brief If parsing attachments, the member posting */ protected ?Member $member = NULL; /** * @brief If parsing attachments, the Editor area we're parsing in. e.g. "core_Signatures". */ protected string|bool|null $area = NULL; /** * @brief Loose Profanity Filters */ protected array $looseProfanity = array(); /** * @brief Exact Profanity Filters */ protected array $exactProfanity = array(); /** * @brief Case-sensitive Acronyms * @var array (array key is the acronym, value is an array with 'a_long' and 'a_type' keys) */ public array $caseSensitiveAcronyms = array(); /** * @brief Case-insensitive Acronyms * @var array (array key is the acronym in lowercase, value is an array with 'a_long' and 'a_type' keys) */ public array $caseInsensitiveAcronyms = array(); /** * @brief If cleaning HTML, the HTMLPurifier object */ protected ?HTMLPurifier $htmlPurifier = NULL; /** * @brief Save on queries and fetch the alt label would just the once */ protected ?string $_altLabelWord = NULL; /** * Constructor * * @param mixed $attachIds array of ID numbers to idenfity content for attachments if the content has been saved - the first two must be int or null, the third must be string or null. If content has not been saved yet, an MD5 hash used to claim attachments after saving. * @param Member|null $member The member posting, NULL will use currently logged in member. * @param bool|string $area If parsing BBCode or attachments, the Editor area we're parsing in. e.g. "core_Signatures". A boolean value will allow or disallow all BBCodes that are dependant on area. * @param bool $filterProfanity Remove profanity? * @param callback|null $htmlPurifierConfig A function which will be passed the HTMLPurifier_Config object to customise it - see example * @param bool $parseAcronyms Parse acronyms? * @param ?int $attachIdsLang Language ID number if this Editor is part of a Translatable field. * @return void */ public function __construct( mixed $attachIds=NULL, Member $member=NULL, bool|string $area=FALSE, bool $filterProfanity=TRUE, callable $htmlPurifierConfig=NULL, bool $parseAcronyms=TRUE, int $attachIdsLang=NULL ) { /* Set the Member */ $this->member = $member ?: Member::loggedIn(); /* Set the member and area */ if ( $attachIds ) { $this->area = $area; } /* Get attachments */ $this->attachIds = $attachIds; $this->attachIdsLang = $attachIdsLang; if( $attachIds !== NULL ) { $where = array( array( 'location_key=?', $area ) ); if ( is_array( $attachIds ) ) { $i = 1; foreach ( $attachIds as $id ) { $where[] = array("id{$i}=?", $id); $i++; } } elseif ( is_string( $attachIds ) ) { $where[] = array( 'temp=?', $attachIds ); } $this->existingAttachments = iterator_to_array( Db::i()->select( '*', 'core_attachments_map', $where )->setKeyField( 'attachment_id' ) ); $this->mappedAttachments = array_keys( $this->existingAttachments ); } /* Get profanity filters */ if ( $filterProfanity ) { foreach( Profanity::getProfanity() AS $profanity ) { if ( $profanity->action == 'swap' ) { if ( $profanity->m_exact ) { $this->exactProfanity[ $profanity->type ] = $profanity->swop; } else { $this->looseProfanity[ $profanity->type ] = $profanity->swop; } } } } /* Get HTMLPurifier Configuration */ if ( !function_exists('idn_to_ascii') ) { IPS::$PSR0Namespaces['TrueBV'] = ROOT_PATH . "/system/3rd_party/php-punycode"; require_once ROOT_PATH . "/system/3rd_party/php-punycode/polyfill.php"; } require_once ROOT_PATH . "/system/3rd_party/HTMLPurifier/HTMLPurifier.auto.php"; $this->htmlPurifier = new HTMLPurifier( $this->_htmlPurifierConfiguration( $htmlPurifierConfig ) ); /* Get acronyms */ if ( $parseAcronyms ) { $this->caseSensitiveAcronyms = iterator_to_array( Db::i()->select( array( 'a_short', 'a_long', 'a_type' ), 'core_acronyms', array( 'a_casesensitive=1' ), 'LENGTH(a_short) DESC' )->setKeyField( 'a_short' ) ); $this->caseInsensitiveAcronyms = array(); foreach ( Db::i()->select( array( 'a_short', 'a_long', 'a_type' ), 'core_acronyms', array( 'a_casesensitive=0' ), 'LENGTH(a_short) DESC' )->setKeyField( 'a_short' ) as $k => $v ) { $this->caseInsensitiveAcronyms[ mb_strtolower( $k ) ] = $v; } } } /** * Parse * * @param string $value HTML to parse * @return string */ public function parse( string $value ): string { /* Clean HTML */ $value = $this->purify( $value ); /* Profanity, etc. */ if ( $value ) { $value = $this->_parseContent( $value ); } // /* Clean HTML */ // $value = $this->purify( $value ); // this is almost certainly redundant now. After the first use of purify, there shouldn't be anything this will remove /* Replace any {fileStore.whatever} tags with */ $value = static::replaceFileStoreTags( $value ); /* HTML Purifier converts
to

. In browsers, this gets rendered as

*/ $value = str_replace( "
", "", $value ); /* Return */ return $value; } /** * Parse * * @param string $value HTML to run through HTML purifier * @return string */ public function purify( string $value ): string { /* Clean HTML */ if ( $value and $this->htmlPurifier ) { $value = $this->htmlPurifier->purify( $value ); } return $value; } /** * Returns the blank image used as a placeholder to facilitate lazy loading * * @return string */ public static function blankImage(): string { return (string) Url::internal( "applications/core/interface/js/spacer.png", 'none', NULL, array(), Url::PROTOCOL_RELATIVE ); } /** * Returns a url to a blank page used as a placeholder to facilitate lazy loading in frames * * @return string */ public static function blankPage(): string { return (string) Url::internal( "applications/core/interface/index.html", 'none', NULL, array(), Url::PROTOCOL_RELATIVE ); } /** * Remove image proxy * * @param string $content HTML to parse * @param bool $useProxyUrl Use the proxied image URL (the locally stored image) instead of the original URL * @return string */ public static function removeImageProxy( string $content, bool $useProxyUrl=FALSE ): string { $source = new DOMDocument( '1.0', 'UTF-8' ); $source->loadHTML( DOMDocument::wrapHtml( $content ) ); /* Get document images */ $contentImages = $source->getElementsByTagName( 'img' ); foreach( $contentImages as $element ) { static::_removeImageProxy($element, $useProxyUrl); } /* Get DOMDocument output */ $content = DOMParser::getDocumentBodyContents( $source ); /* Replace file storage tags */ $content = preg_replace( '/<fileStore\.([\d\w\_]+?)>/i', '', $content ); /* DOMDocument::saveHTML will encode the base_url brackets, so we need to make sure it's in the expected format. */ return str_replace( '<___base_url___>', '<___base_url___>', $content ); } /** * Parse content to remove old lazy loading * * @param string $content HTML to parse * @return string */ public static function parseLazyLoad( string $content ): string { /* Lazy loading applies to images, iframes and videos - return now if we don't detect any with basic string checks */ if( mb_strpos( $content, 'loadHTML( DOMDocument::wrapHtml( $content ) ); /* Swap data-src for src */ $contentImages = $source->getElementsByTagName( 'img' ); foreach( $contentImages as $element ) { if ( $element->hasAttribute('data-src') ) { $element->setAttribute( 'src', $element->getAttribute('data-src') ); $element->removeAttribute( 'data-src' ); } $element->setAttribute( 'loading', 'lazy' ); /* Convert ratio into height */ if( !$element->hasAttribute( 'height' ) AND $element->hasAttribute( 'width' ) AND $element->hasAttribute( 'data-ratio' ) ) { $element->setAttribute( 'height', ( (int) $element->getAttribute( 'width' ) / 100 ) * (int) $element->getAttribute( 'data-ratio' ) ); $element->removeAttribute( 'data-ratio' ); } } $contentVideos = $source->getElementsByTagName( 'video' ); foreach( $contentVideos as $element ) { if ( $element->hasAttribute('data-video-embed') ) { $element->setAttribute( 'data-controller', 'core.global.core.embeddedvideo' ); $element->removeAttribute( 'data-video-embed' ); } $element->setAttribute( 'preload', 'metadata' ); } /* Swap data-video-src for src */ $contentVideos = $source->getElementsByTagName( 'source' ); foreach( $contentVideos as $element ) { if ( $element->parentNode->tagName === 'video' and $element->hasAttribute('data-video-src') ) { $element->setAttribute( 'src', $element->getAttribute('data-video-src') ); $element->removeAttribute( 'data-video-src' ); } } /* Fix Audio Tags */ $contentAudio = $source->getElementsByTagName( 'audio' ); foreach( $contentAudio as $element ) { if ( $element->hasAttribute('data-audio-embed') ) { $element->setAttribute( 'data-controller', 'core.global.core.embeddedaudio' ); $element->removeAttribute( 'data-audio-embed' ); } $element->setAttribute( 'preload', 'metadata' ); } /* Swap data-embed-src for src */ $contentEmbeds = $source->getElementsByTagName( 'iframe' ); foreach( $contentEmbeds as $element ) { if ( $element->hasAttribute('data-embed-src') ) { $element->setAttribute( 'src', $element->getAttribute('data-embed-src') ); $element->removeAttribute( 'data-embed-src' ); } $element->setAttribute( 'loading', 'lazy' ); } /* Get DOMDocument output */ $content = DOMParser::getDocumentBodyContents( $source ); /* Replace file storage tags */ $content = preg_replace( '/<fileStore\.([\d\w\_]+?)>/i', '', $content ); /* DOMDocument::saveHTML will encode the base_url brackets, so we need to make sure it's in the expected format. */ return str_replace( '<___base_url___>', '<___base_url___>', $content ); } /** * Replace {fileStore.xxx} with * * @param string $value HTML to parse * @return string */ public static function replaceFileStoreTags( string $value ): string { /* Some tags have multiple __base_url__ replacements, so we have to replace this in a safe way ensuring we only match inside A, IMG, IFRAME and VIDEO tags to prevent tampering */ preg_match_all( '#<(img|a|iframe|video|audio|source|blockquote)([^>]+?)%7B___base_url___%7D([^>]+?)>#i', $value, $matches, PREG_SET_ORDER ); foreach( $matches as $val ) { $changed = $val[0]; /* srcset can have multiple urls in it */ preg_match( '#srcset=(\'|")([^\'"]+?)(\1)#i', $changed, $srcsetMatches ); if ( isset( $srcsetMatches[2] ) ) { if ( mb_stristr( $srcsetMatches[2], '%7B___base_url___%7D' ) ) { $changed = str_replace( $srcsetMatches[2], str_replace( '%7B___base_url___%7D', '<___base_url___>', $srcsetMatches[2] ), $changed ); } } $changed = preg_replace( '#(href|src|data\-fileid|data\-ipshover\-target|cite)=(\'|")%7B___base_url___%7D/#i', '\1=\2<___base_url___>/', $changed ); if ( $changed != $val[0] ) { $value = str_replace( $val[0], $changed, $value ); } } /* Replace {fileStore.xxx} with */ $value = preg_replace( '#(srcset|src|href|cite)=(\'|")(%7B|\{)fileStore\.([\d\w\_]+?)(%7D|\})/#i', '\1=\2/', $value ); /* Return */ return $value; } /* !Parser: HTMLPurifier */ /** * Get HTML Purifier Configuration * * @param callback|null $callback A function which will be passed the HTMLPurifier_Config object to customise it * @return HTMLPurifier_Config */ protected function _htmlPurifierConfiguration( callable $callback = NULL ): HTMLPurifier_Config { /* Start with a base configruation */ $config = HTMLPurifier_Config::createDefault(); /* HTMLPurifier by default caches data to disk which we cannot allow. Register our custom cache definiton to use \IPS\Data\Store instead */ $definitionCacheFactory = HTMLPurifier_DefinitionCacheFactory::instance(); $definitionCacheFactory->register( 'IPSCache', "HtmlPurifierDefinitionCache" ); require_once( ROOT_PATH . '/system/Text/HtmlPurifierDefinitionCache.php' ); $config->set( 'Cache.DefinitionImpl', 'IPSCache' ); /* Allow iFrames from services we allow. We limit this to a whitelist because to allow any iframe would open us to phishing and other such security issues */ $config->set( 'HTML.SafeIframe', true ); $config->set( 'URI.SafeIframeRegexp', static::safeIframeRegexp() ); $config->set( 'Output.Newline', "\n" ); /* Set allowed CSS classes. We limit this to a whitelist because to allow any iframe would open us to phishing (for example, someone posts something which, by using our CSS classes, looks like a login form), and general annoyances */ $config->set( 'Attr.AllowedClasses', static::getAllowedCssClasses() ); /* Increase default image width */ $config->set( 'CSS.MaxImgLength', "4800px" ); $config->set( 'HTML.MaxImgLength', "4800" ); /* Callback */ if ( $callback ) { $callback( $config ); } /* HTML Definition */ $htmlDefinition = $config->getHTMLDefinition( TRUE ); $this->_htmlPurifierModifyHtmlDefinition( $htmlDefinition ); /* CSS Definition */ $cssDefinition = $config->getCSSDefinition(); $this->_htmlPurifierModifyCssDefinition( $cssDefinition, $config ); $uri = $config->getDefinition('URI'); $uri->addFilter( new HtmlPurifierHttpsImages(), $config ); /* Return */ return $config; } /** * Customize HTML Purifier HTML Definition * * @param HTMLPurifier_HTMLDefinition $def The definition * @return void */ protected function _htmlPurifierModifyHtmlDefinition( HTMLPurifier_HTMLDefinition $def ) : void { /* Links (set by _parseAElement) */ $def->addAttribute( 'a', 'rel', 'Text' ); /* srcset for emoticons (used by _parseImgElement) */ $def->addAttribute( 'img', 'srcset', new HtmlPurifierSrcsetDef( TRUE ) ); /* Quotes (used by ipsquote editor plugin) */ $def->addAttribute( 'blockquote', 'data-ipsquote', 'Bool' ); $def->addAttribute( 'blockquote', 'data-ipsquote-timestamp', 'Number' ); $def->addAttribute( 'blockquote', 'data-ipsquote-username', 'Text' ); $def->addAttribute( 'blockquote', 'data-ipsquote-contentapp', 'Text' ); $def->addAttribute( 'blockquote', 'data-ipsquote-contentclass', 'Text' ); $def->addAttribute( 'blockquote', 'data-ipsquote-contenttype', 'Text' ); $def->addAttribute( 'blockquote', 'data-ipsquote-contentid', 'Number' ); $def->addAttribute( 'blockquote', 'data-ipsquote-contentcommentid', 'Number' ); $def->addAttribute( 'blockquote', 'data-ipsquote-userid', 'Number' ); $def->addAttribute( 'blockquote', 'data-cite', 'Text' ); $def->addAttribute( 'blockquote', 'cite', 'Text' ); $def->addAttribute( 'div', 'data-ipstruncate', 'Text' ); // this is used on the div.ipsQuote_contents element inside the blockquote $def->addElement( "header", 'Block', "Inline", 'Common' ); /* Mentions (used by ipsmentions editor plugin) */ $def->addAttribute( 'a', 'data-ipshover', new HtmlPurifierSwitchAttrDef( 'a', array( 'data-ipshover-target' ), new HTMLPurifier_AttrDef_HTML_Bool(''), new HTMLPurifier_AttrDef_Enum( array() ) ) ); $def->addAttribute( 'a', 'data-ipshover-target', new HtmlPurifierInternalLinkDef( TRUE, array( array( 'app' => 'core', 'module' => 'members', 'controller' => 'profile', 'do' => 'hovercard' ) ) ) ); $def->addAttribute( 'a', 'data-mentionid', 'Number' ); $def->addAttribute( 'a', 'contenteditable', 'Enum#false' ); /* Emoticons (used by the ipsautolink plugin) */ $def->addAttribute( 'img', 'data-emoticon', 'Bool' ); // Identifies emoticons and stops lightbox running on them /* Attachments (set by _parseAElement, _parseImgElement and "insert existing attachment") - Gallery/Downloads use the full URL rather than an ID, hence Text */ $def->addAttribute( 'a', 'data-fileid', new HtmlPurifierIntOrInternalLink() ); $def->addAttribute( 'img', 'data-fileid', new HtmlPurifierIntOrInternalLink() ); $def->addAttribute( 'img', 'data-full-image', 'Text' ); $def->addAttribute( 'a', 'data-fileext', 'Text' ); /* Existing media (inserted with data-extension by the JS so that _getFile is able to locate) */ $def->addAttribute( 'img', 'data-extension', 'Text' ); $def->addAttribute( 'a', 'data-extension', 'Text' ); /* This is needed because a tags can have images whose width is defined by the image; Attachments are a good example of this */ $def->addAttribute( 'a', 'style', 'Text' ); /* Lazy loading */ $def->addAttribute( 'img', 'width', 'Number' ); $def->addAttribute( 'img', 'height', 'Number' ); $def->addAttribute( 'img', 'style', 'Text' ); $def->addAttribute( 'iframe', 'style', 'Text' ); $def->addAttribute( 'div', 'style', 'Text' ); $def->addAttribute( 'img', 'loading', new HTMLPurifier_AttrDef_Enum( [ 'lazy', 'eager' ] ) ); $def->addAttribute( 'iframe', 'data-embed-src', new HTMLPurifier_AttrDef_URI( TRUE ) ); $def->addAttribute( 'iframe', 'src', 'Text' ); /* iFrames (used by embeddableMedia) */ $def->addAttribute( 'iframe', 'data-controller', new HTMLPurifier_AttrDef_Enum( array( 'core.front.core.autosizeiframe' ) ) ); // used in core/global/embed/iframe.phtml $def->addAttribute( 'iframe', 'data-embedid', 'Text' ); // used in core/global/embed/iframe.phtml $def->addAttribute( 'iframe', 'data-embedauthorid', 'Number' ); // used for embed notifications $def->addAttribute( 'iframe', 'data-embedcontent', 'Text' ); // used in embeddableMedia $def->addAttribute( 'iframe', 'data-ipsembed-contentapp', 'Text' ); // used in embeddableMedia $def->addAttribute( 'iframe', 'data-ipsembed-contentid', 'Text' ); // used in embeddableMedia $def->addAttribute( 'iframe', 'data-ipsembed-contentclass', 'Text' ); // used in embeddableMedia $def->addAttribute( 'iframe', 'data-ipsembed-contentcommentid', 'Text' ); // used in embeddableMedia $def->addAttribute( 'iframe', 'data-ipsembed-timestamp', 'Text' ); // used in embeddableMedia $def->addAttribute( 'iframe', 'data-internalembed', 'Text' ); // used in embeddableMedia $def->addAttribute( 'iframe', 'allowfullscreen', 'Text' ); // Some services will specify this property $def->addAttribute( 'iframe', 'allow', 'Text' ); // Some services will specify this property. We replace it with individual allow* attributes later on in the parsing process /* Custom Iframes need these */ $def->addAttribute( 'iframe', 'width', 'Number' ); // Some services will specify this property $def->addAttribute( 'iframe', 'height', 'Number' ); // Some services will specify this property /* Tiptap Adds Style and Classes to Spans */ $def->addAttribute( 'span', 'class', 'Text' ); $def->addAttribute( 'span', 'style', 'Text' ); $def->addAttribute( 'span', 'data-ips-font-size', 'Text' ); /* Highlight colors */ $def->addElement( "mark", "Inline", "Inline", "Common" ); $def->addAttribute( 'mark', 'data-i-background-color', 'Text' ); /* data-controllers */ $allowedDivDataControllers = array( 'core.front.core.articlePages', // [page] (set by _parseContent) ); foreach( Application::enabledApplications() as $app ) { $settingsFile = $app->getApplicationPath() . "/data/parser.json"; if( file_exists( $settingsFile ) ) { $contents = json_decode( file_get_contents( $settingsFile ), true ); $allowedDivDataControllers = array_merge( $allowedDivDataControllers, $contents['controllers'] ); } } $def->addAttribute( 'div', 'data-controller', new HTMLPurifier_AttrDef_Enum( $allowedDivDataControllers, TRUE ) ); /* [page] (set by _parseContent) */ $def->addAttribute( 'div', 'data-role', new HTMLPurifier_AttrDef_Enum( array( 'contentPage' ), TRUE ) ); $def->addAttribute( 'hr', 'data-role', new HTMLPurifier_AttrDef_Enum( array( 'contentPageBreak' ), TRUE ) ); /* data-munge-src used by _removeMunge() */ $def->addAttribute( 'img', 'data-munge-src', 'Text' ); $def->addAttribute( 'iframe', 'data-munge-src', 'Text' ); /* Videos */ $def->addElement( 'video', 'Inline', 'Optional: (source, Flow) | (Flow, source) | Flow', 'Common', array( 'controls' => 'Bool', 'data-controller' => new HTMLPurifier_AttrDef_Enum( array( 'core.global.core.embeddedvideo' ) ), 'data-video-embed' => 'Text', 'preload' => new HTMLPurifier_AttrDef_Enum( [ 'auto', 'metadata', 'none' ] ), 'style' => 'Text' ) ); $def->addAttribute( 'video', 'data-video-preview-time', 'Number' ); $def->addElement( 'source', 'Inline', 'Flow', 'Common', array( 'src' => new HTMLPurifier_AttrDef_URI( TRUE ), 'srcset' => new HtmlPurifierSrcsetDef( TRUE ), 'media' => 'Text', 'type' => 'Text', 'data-video-src' => new HTMLPurifier_AttrDef_URI( TRUE ) ) ); /* Audio */ $def->addElement('audio', 'Inline', 'Optional: (source, Flow) | (Flow, source) | Flow', 'Common', array( 'controls' => 'Bool', 'data-controller' => new HTMLPurifier_AttrDef_Enum( array( 'core.global.core.embeddedaudio' ) ), 'src' => new HTMLPurifier_AttrDef_URI( TRUE ), 'srcset' => new HtmlPurifierSrcsetDef( TRUE ), 'media' => 'Text', 'type' => 'Text', 'data-audio-embed' => 'Text', 'data-audio-src' => new HTMLPurifier_AttrDef_URI( TRUE ), 'preload' => new HTMLPurifier_AttrDef_Enum( [ 'auto', 'metadata', 'none' ] ) )); /* Picture tag - we don't use it, but RSS imports might */ $def->addElement( 'picture', 'Inline', 'Optional: (source, Flow) | (Flow, source) | Flow', 'Common', array() ); /* Tiptap CodeboxLowLight Styles */ $def->addAttribute( 'code', 'class', 'Text' ); $def->addAttribute( 'pre', 'data-language', 'Text' ); $def->addAttribute( 'pre', 'spellcheck', 'Bool' ); /* Font colors */ $def->addAttribute( 'span', 'data-i-color', 'Text' ); $def->addAttribute( 'span', 'data-i-background-color', 'Text' ); /* Tables */ $def->addAttribute( 'table', 'style', 'Text' ); $def->addAttribute( 'tr', 'style', 'Text' ); $def->addAttribute( 'th', 'style', 'Text' ); $def->addAttribute( 'td', 'style', 'Text' ); /* This is important for the tiptap og embed embeds */ $def->addElement( 'figure', 'Block', 'Optional: Flow | (source, Flow) | (Flow, source)', 'Common', array( 'class'=> 'Text' ) ); $def->addElement( 'figcaption', 'Block', 'Optional: Flow | (source, Flow) | (Flow, source)', 'Common', array( 'class' => 'Text' )); foreach ( static::$ogFields as $field ) { $def->addAttribute( 'figure', 'data-og-' . $field, 'Text' ); } foreach ( ['figure','div','video','audio','img','iframe','embed'] as $embedType ) { $def->addAttribute( $embedType, 'data-og-user_text', 'Text' ); } /* IPS Boxes */ $def->addElement( 'details', 'Block', 'Optional: Flow | (source, Flow) | (Flow, source)', 'Common' ); $def->addElement( 'summary', 'Block', "Optional: Flow | (source, Flow) | (Flow, source)", 'Common' ); $def->addAttribute('details', 'class', 'Text'); $def->addAttribute( 'summary', 'data-i-background-color', 'Text' ); $def->addAttribute( 'details', 'data-i-background-color', 'Text' ); $def->addAttribute( 'div', 'data-i-background-color', 'Text' ); $def->addAttribute( 'details', 'open', 'Text' ); /* I tag for fa icons */ $def->addElement( 'i', 'Inline', "Inline", "Common" ); $def->addAttribute( 'i', 'class', 'Text' ); /* Allow BR tags */ $def->addElement( 'br', 'Inline', "Inline", "Common" ); Bridge::i()->parserModifyHTMLDefinintion( $def ); } /** * @var string[] The allowed fields in an embed parsed from OG data. */ public static array $ogFields = [ 'url', 'site_name', 'title', 'description', 'type', 'image', 'image_width', 'image_height', 'locale', 'favicon_url' ]; /** * @brief Maximum allowed width/height px values (to prevent causing page layout oddities) */ protected int $cssMaxWidthHeight = 1000; /** * @brief Maximum allowed border width px value (to prevent causing page layout oddities) */ protected int $cssMaxBorderWidth = 50; /** * Customize HTML Purifier CSS Definition * * @param HTMLPurifier_CSSDefinition $def The definition * @param HTMLPurifier_Config $config HTML Purifier configuration object * @return void */ protected function _htmlPurifierModifyCssDefinition( HTMLPurifier_CSSDefinition $def, HTMLPurifier_Config $config ) : void { /* Do not allow negative margins */ $margin = $def->info['margin-right'] = $def->info['margin-left'] = $def->info['margin-bottom'] = $def->info['margin-top'] = new HTMLPurifier_AttrDef_CSS_Composite( array( new HTMLPurifier_AttrDef_CSS_Length( 0 ), new HTMLPurifier_AttrDef_CSS_Percentage( TRUE ), new HTMLPurifier_AttrDef_Enum(array('auto')) ) ); $def->info['margin'] = new HTMLPurifier_AttrDef_CSS_Multiple( $margin ); /* Don't allow white-space:nowrap */ $def->info['white-space'] = new HTMLPurifier_AttrDef_Enum( array( 'normal', 'pre', 'pre-wrap', 'pre-line') ); /* Limit the maximum width and height allowed */ $def->info['width'] = $def->info['height'] = new HTMLPurifier_AttrDef_CSS_Composite( array( new HTMLPurifier_AttrDef_CSS_Length( '0px', $this->cssMaxWidthHeight . 'px' ), new HTMLPurifier_AttrDef_CSS_Percentage( true ), new HTMLPurifier_AttrDef_Enum( array( 'auto' ) ) ) ); /* Limit the maximum border width allowed */ $border_width = $def->info['border-top-width'] = $def->info['border-bottom-width'] = $def->info['border-left-width'] = $def->info['border-right-width'] = new HTMLPurifier_AttrDef_CSS_Composite(array( new HTMLPurifier_AttrDef_Enum( array( 'thin', 'medium', 'thick' ) ), new HTMLPurifier_AttrDef_CSS_Length( '0px', $this->cssMaxBorderWidth . 'px' ) ) ); $def->info['border-width'] = new HTMLPurifier_AttrDef_CSS_Multiple( $border_width ); /* We have to reset this so the constructor picks up the new values we just specified */ $def->info['border'] = $def->info['border-bottom'] = $def->info['border-top'] = $def->info['border-left'] = $def->info['border-right'] = new HTMLPurifier_AttrDef_CSS_Border( $config ); } /** * Get URL bases (whout schema) that we'll allow iframes from * * @return array|string */ protected static function safeIframeRegexp(): array|string { $return = array(); /* 3rd party sites (YouTube, etc.) */ foreach ( static::allowedIFrameBases() as $base ) { $return[] = '(https?:)?//' . preg_quote( $base, '%' ); } /* Some, but not all local URLs Allowed: Any URLs which go through the front-end, e.g.: site.com/?app=core&module=system&controller=embed&url=whatever site.com/index.php?app=core&module=system&controller=embed&url=whatever site.com/topic/1-test/?do=embed site.com/index.php?/topic/1-test/?do=embed site.com/index.php?app=forums&module=forums&controller=topic&id=1&do=embed Not Allowed: Anything which goes to anything in an /interface directory - e.g.: site.com/applications/core/interface/file/attachment.php - this would automatically cause files to be downloaded Not Allowed: Any file traversal to expose FileSystem directories, e.g: site.com/applications/core/../uploads/monthly_xx_xx/file.js? Not Allowed: URLs to the open proxy: site.com/index.php?app=core&module=system&controller=redirect */ $notAllowed = array(); foreach( Application::enabledApplications() as $app ) { $notAllowed[] = str_replace( '/', '(?:/{1,})', '(?:/{0,})' . preg_quote( 'applications/' . $app->directory . '/interface/', '%' ) ); } foreach( File::getStore() as $configuration ) { if ( $configuration['method'] == 'FileSystem' and ! empty( $configuration['configuration'] ) and $config = json_decode( $configuration['configuration'], TRUE ) ) { if ( !empty( $config['dir'] ) and mb_strpos( $config['dir'], '{root}' ) !== false ) { if ( $path = trim( str_replace( '{root}', '', $config['dir'] ), '\/' ) ) { $notAllowed[] = '(?:.+?/{1,})\.{1,}(?:/{1,})' . $path; } } } } $return[] = '(https?:)?//' . preg_quote( str_replace( array( 'http://', 'https://' ), '', Settings::i()->base_url ), '%' ) . '\/?(\?|index\.php\?|(?!' . implode( '|', $notAllowed ) . ').+?\?)((?!(controller|section)=redirect).)*$'; $return[] = preg_quote( '%7B___base_url___%7D', '%' ) . '\/?(\?|index\.php\?|(?!' . implode( '|', $notAllowed ) . ').+?\?)((?!(controller|section)=redirect).)*$'; /* Return */ return '%^(' . implode( '|', $return ) . ')%'; } /** * Get URL bases (whout schema) that we'll allow iframes from * * @return array */ protected static function allowedIFrameBases(): array { $return = array(); /* Our default embed options */ $return = array_merge( $return, array( 'www.youtube.com/embed/', 'www.youtube-nocookie.com/embed/', 'player.vimeo.com/video/', 'www.hulu.com/embed.html', 'www.collegehumor.com/e/', 'embed-ssl.ted.com/', 'embed.ted.com', 'embed.spotify.com/', 'www.dailymotion.com/embed/', 'www.funnyordie.com/', 'coub.com/', 'www.reverbnation.com/', 'api.smugmug.com/services/embed/', 'www.google.com/maps/', 'www.screencast.com/users/', 'fast.wistia.net/embed/', 'www.screencast.com/users/', 'players.brightcove.net/', ) ); /* Extra admin-defined options */ foreach( Application::enabledApplications() as $app ) { $settingsFile = $app->getApplicationPath() . "/data/parser.json"; if( file_exists( $settingsFile ) ) { $contents = json_decode( file_get_contents( $settingsFile ), true ); $return = array_merge( $return, $contents['iframe'] ); } } /* If the CMS root URL is not inside the IPS4 directory, then embeds will fails as the src will not be allowed */ if ( Application::appIsEnabled( 'cms' ) and Settings::i()->cms_root_page_url ) { $pages = iterator_to_array( Db::i()->select( 'database_page_id', 'cms_databases', array( 'database_page_id > 0' ) ) ); foreach ( new ActiveRecordIterator( Db::i()->select( '*', 'cms_pages', array( Db::i()->in( 'page_id', $pages ) ) ), 'IPS\cms\Pages\Page' ) as $page ) { $return[] = str_replace( array( 'http://', 'https://' ), '', $page->url() ); } } return $return; } /** * Get allowed CSS classes * * @return array */ protected function getAllowedCssClasses(): array { /* Init */ $return = array(); /* Quotes (used by ipsquote editor plugin) */ $return[] = 'ipsQuote'; $return[] = 'ipsQuote_citation'; $return[] = 'ipsQuote_contents'; /* Code (used by ipscode editor plugin) */ $return[] = 'ipsCode'; $return[] = 'prettyprint'; $return[] = 'prettyprinted'; $return[] = 'lang-auto'; $return[] = 'lang-javascript'; $return[] = 'lang-php'; $return[] = 'lang-css'; $return[] = 'lang-html'; $return[] = 'lang-xml'; $return[] = 'lang-c'; $return[] = 'lang-sql'; $return[] = 'lang-lua'; $return[] = 'lang-swift'; $return[] = 'lang-perl'; $return[] = 'lang-python'; $return[] = 'lang-ruby'; $return[] = 'lang-latex'; $return[] = 'tag'; $return[] = 'pln'; $return[] = 'atn'; $return[] = 'atv'; $return[] = 'pun'; $return[] = 'com'; $return[] = 'kwd'; $return[] = 'str'; $return[] = 'lit'; $return[] = 'typ'; $return[] = 'dec'; $return[] = 'src'; $return[] = 'nocode'; /* Box */ $return[] = 'ipsRichTextBox'; $return[] = "ipsRichTextBox--alwaysopen"; $return[] = "ipsRichTextBox--expandable"; $return[] = "ipsRichTextBox--collapsible"; $return[] = 'ipsRichTextBox__title'; /* Alignment settings */ $return[] = 'ipsRichText__align'; $return[] = 'ipsRichText__align--left'; $return[] = 'ipsRichText__align--right'; $return[] = 'ipsRichText__align--block'; $return[] = 'ipsRichText__align--inline'; $return[] = 'ipsRichText__align--width-small'; $return[] = 'ipsRichText__align--width-big'; $return[] = 'ipsRichText__align--width-medium'; $return[] = 'ipsRichText__align--width-fullwidth'; $return[] = 'ipsRichText__align--width-custom'; /* Images and attachments (used when attachments are inserted into the editor) */ $return[] = 'ipsImage'; $return[] = 'ipsImage_thumbnailed'; $return[] = 'ipsAttachLink'; $return[] = 'ipsAttachLink_image'; $return[] = 'ipsAttachLink_left'; $return[] = 'ipsAttachLink_right'; $return[] = 'ipsEmoji'; /* Embeds (used by various return values of embeddedMedia) */ $return[] = 'ipsEmbedded'; $return[] = 'ipsEmbeddedVideo'; $return[] = 'ipsEmbeddedVideo_limited'; $return[] = 'ipsEmbeddedOther'; $return[] = 'ipsEmbeddedOther--google-maps'; $return[] = 'ipsEmbeddedOther--iframely'; $return[] = 'iframely-embed'; $return[] = 'iframely-player'; $return[] = 'iframely-responsive'; $return[] = 'ipsEmbeddedOther_limited'; $return[] = 'ipsRawIframe'; $return[] = 'ipsEmbedded__wrap'; $return[] = 'ipsEmbedded__wrap--center'; $return[] = 'ipsEmbedded__wrap--end'; /* Links (Used to replace disallowed URLs */ $return[] = 'ipsType_noLinkStyling'; $return[] = 'ipsMention'; /* Brightcove */ $return[] = 'ipsEmbeddedBrightcove'; $return[] = 'ipsEmbeddedBrightcove_inner'; $return[] = 'ipsEmbeddedBrightcove_frame'; /* OG Embed */ $return[] = 'ipsEmbedded_og'; foreach ( [ 'image', 'description', 'title', 'title--alone', 'site-name', 'favicon'] as $k ) { $return[] = 'ipsEmbedded_og__' . $k; } /* Tables */ $return[] = "ipsRichText__table-wrapper"; /* Custom */ foreach( Application::enabledApplications() as $app ) { $settingsFile = $app->getApplicationPath() . "/data/parser.json"; if( file_exists( $settingsFile ) ) { $contents = json_decode( file_get_contents( $settingsFile ), true ); $return = array_merge( $return, $contents['css'] ); } } /* Language classes */ foreach( json_decode( file_get_contents( Application::load( 'core' )->getApplicationPath() . '/data/allKnownCodeLanguages.json' ), true )['languages'] as $supportedLang ) { $return[] = strtolower( "language-{$supportedLang}" ); } return $return; } /* !Parser: Main Parser */ /** * @brief Does the content contain [page] tags? */ protected bool $containsPageTags = FALSE; /** * @brief Open tags */ protected array $openAbbrTags = array(); /** * Parse Profanity, etc. by loading into a DOMDocument * * @param string $value HTML to parse * @return string */ protected function _parseContent( string $value ): string { /* The editor button just drops in a
, so we need to make sure that the structure is correct, as follows:

Page one


Page two

The editor will just have

Page one


Page two

*/ /* Parse */ $parser = new DOMParser( array( $this, '_parseDomElement' ), array( $this, '_parseDomText' ) ); $document = $parser->parseValueIntoDocument( $value ); /* Return */ return DOMParser::getDocumentBodyContents( $document ); } /** * Parse HTML element (e.g. ,

, , etc.) * * @param DOMElement $element The element from the source document to parse * @param DOMNode $parent The node from the new document which will be this node's parent * @param DOMParser $parser DOMParser Object * @return void */ public function _parseDomElement(DOMElement $element, DOMNode $parent, DOMParser $parser ) : void { /* Start of an ? */ $okayToParse = TRUE; if ( $element->tagName === 'abbr' and $element->hasAttribute('title') ) { $title = $element->getAttribute('title'); if ( !in_array( $title, $this->openAbbrTags ) ) { $this->openAbbrTags[] = $title; } else { $okayToParse = FALSE; } } /* Import */ if ( $okayToParse ) { /* Import the element as it is */ $ownerDocument = $parent->ownerDocument ?: $parent; $newElement = $ownerDocument->importNode( $element ); /* Element-specific parsing */ $newElement = $this->_parseElement( $newElement ); /* Append */ if ( $newElement instanceof DOMElement ) { $parent->appendChild( $newElement ); } /* Swap out emoticons that should be plaintext (meaning we hit the maximum limit of emoticons per editor) */ foreach( $parent->getElementsByTagName( 'img' ) AS $img ) { if ( $img->hasAttribute( 'data-ipsEmoticon-plain' ) ) { $replace = $parent->appendChild( new DOMText( $img->getAttribute( 'data-ipsEmoticon-plain' ) ) ); $parent->replaceChild( $replace, $img ); } } } else { $newElement = $parent; } // we only care to copy the element's children if it's new if ( $newElement instanceof DOMNode ) { $parser->_parseDomNodeList( $element->childNodes, $newElement ); } /* Finish */ if ( $okayToParse ) { /* End of an ? */ if ( $element->tagName === 'abbr' and $element->hasAttribute('title') ) { $k = array_search( $element->getAttribute('title'), $this->openAbbrTags ); if ( $k !== FALSE ) { unset( $this->openAbbrTags[ $k ] ); } } /* If we did have children, but now we don't, drop this element to avoid unintentional whitespace */ if ( $newElement instanceof DOMNode and $newElement->parentNode and $element->childNodes->length and !$newElement->childNodes->length ) { $parent->removeChild( $newElement ); } } } /** * Parse Text * * @param DOMText $textNode The text from the source document to parse * @param DOMNode $parent The node from the new document which will be this node's parent - passed by reference and may be modified for siblings * @param DOMParser $parser * @return void */ public function _parseDomText(DOMText $textNode, DOMNode &$parent, DOMParser $parser ) : void { /* Init */ $text = $textNode->wholeText; $breakPoints = array( '(' . static::getEmojiRegex() . ')' ); /* Contains [page] tags? */ if ( mb_strpos( $text, '[page]' ) !== FALSE ) { $this->containsPageTags = TRUE; } /* If we have any acronyms, they also need to be breakpoints */ if ( count( $this->caseSensitiveAcronyms ) or count( $this->caseInsensitiveAcronyms ) ) { $breakPoints[] = '((?=<^|\b|\W)(?:' . implode( '|', array_merge( array_map( function ( $value ) { return preg_quote( $value, '/' ); }, array_keys( $this->caseSensitiveAcronyms ) ), array_map( function ( $value ) { return preg_quote( $value, '/' ); }, array_keys( $this->caseInsensitiveAcronyms ) ) ) ) . ')(?=\b|\W|$))'; } /* Loop through each section */ if ( count( $breakPoints ) ) { $sections = array_values( array_filter( preg_split( '/' . implode( '|', $breakPoints ) . '/iu', $text, null, PREG_SPLIT_DELIM_CAPTURE ), function( $val ) { return $val !== ''; } ) ); foreach( $sections as $sectionId => $section ) { $this->_parseTextSection( $section, $parent ); } } else { $this->_parseTextSection( $textNode->wholeText, $parent ); } } /** * Parse a section of text after it has been split into relevant sections * * @param string $section The text from the source document to parse * @param DOMNode $parent The node from the new document which will be this node's parent - passed by reference and may be modified for siblings * @return void */ protected function _parseTextSection( string $section, DOMNode &$parent ) : void { /* If it's empty, skip it */ if ( $section === '' ) { return; } /* HTMLPurifier will strip carrage returns, but if HTML posting is enabled this doesn't happen which leaves blank spaces - so we need to strip here */ if ( !$this->htmlPurifier ) { $section = str_replace( "\r", '', $section ); } /* Profanity */ foreach ( $this->exactProfanity as $bad => $good ) { $section = preg_replace( '/(^|\b|\s)' . preg_quote( $bad, '/' ) . '(\b|\s|!|\?|\.|,|$)/iu', "\\1" . $good . "\\2", $section ); } $section = str_ireplace( array_keys( $this->looseProfanity ), array_values( $this->looseProfanity ), $section ); /* Note what $parent is */ $originalParent = $parent; /* Acronym? */ if ( array_key_exists( $section, $this->caseSensitiveAcronyms ) and !in_array( $this->caseSensitiveAcronyms[ $section ], $this->openAbbrTags ) ) { switch( $this->caseSensitiveAcronyms[ $section ]['a_type'] ) { case 'acronym': $parent = $parent->appendChild( new DOMElement( 'abbr' ) ); $parent->setAttribute( 'title', $this->caseSensitiveAcronyms[ $section ]['a_long'] ); break; case 'link': $replace = ( $parent->tagName != 'a' ); $parentNode = $parent; while( ( $parentNode = $parentNode->parentNode ) !== NULL ) { if( $parentNode instanceof DOMElement AND $parentNode->tagName == 'a' ) { $replace = FALSE; break; } } if( $replace ) { $parent = $parent->appendChild( new DOMElement( 'a' ) ); $parent->setAttribute( 'href', $this->caseSensitiveAcronyms[ $section ]['a_long'] ); try { $rels = $this->_getRelAttributes( Url::createFromString( $this->caseSensitiveAcronyms[ $section ]['a_long'] ) ); } catch(Url\Exception $e ) { $rels = array(); } /* Add rels */ $parent->setAttribute( 'rel', implode( ' ', $rels ) ); } break; } } elseif ( array_key_exists( mb_strtolower( $section ), $this->caseInsensitiveAcronyms ) and !in_array( $this->caseInsensitiveAcronyms[ mb_strtolower( $section ) ], $this->openAbbrTags ) ) { switch( $this->caseInsensitiveAcronyms[ mb_strtolower( $section ) ]['a_type'] ) { case 'acronym': $parent = $parent->appendChild( new DOMElement( 'abbr' ) ); $parent->setAttribute( 'title', $this->caseInsensitiveAcronyms[ mb_strtolower( $section ) ]['a_long'] ); break; case 'link': $replace = ( $parent->tagName != 'a' ); $parentNode = $parent; while( ( $parentNode = $parentNode->parentNode ) !== NULL ) { if( $parentNode instanceof DOMElement AND $parentNode->tagName == 'a' ) { $replace = FALSE; break; } } if( $replace ) { $parent = $parent->appendChild( new DOMElement( 'a' ) ); $parent->setAttribute( 'href', $this->caseInsensitiveAcronyms[ mb_strtolower( $section ) ]['a_long'] ); try { $rels = $this->_getRelAttributes( Url::createFromString( $this->caseInsensitiveAcronyms[ mb_strtolower( $section ) ]['a_long'] ) ); } catch(Url\Exception $e ) { $rels = array(); } /* Add rels */ $parent->setAttribute( 'rel', implode( ' ', $rels ) ); } break; } } /* Emoji? */ if ( ( !$originalParent->getAttribute('class') or !in_array( 'ipsEmoji', explode( ' ', $originalParent->getAttribute('class') ) ) ) and preg_match( '/' . static::getEmojiRegex() . '/u', $section ) ) { $parent = $parent->appendChild( new DOMElement( 'span' ) ); $parent->setAttribute( 'class', 'ipsEmoji' ); } /* Check for emails */ if ( Settings::i()->email_filter_action == 'replace' and preg_match( '/' . static::EMAIL_REGEX . '/u', $section ) ) { $section = preg_replace( '/' . static::EMAIL_REGEX . '/u', Settings::i()->email_filter_replace_text, $section ); } /* Insert the text */ $parent->appendChild( new DOMText( $section ) ); /* Restore the parent */ $parent = $originalParent; } /* !Parser: Element-Specific Parsing */ /** * Element-Specific Parsing * * @param DOMElement $element The element * * @note _parseDomElement() creates a new element and imports it into the document. You can inspect $originalElement if you need to check context. * @return DOMNode|bool|DOMElement|null */ protected function _parseElement( DOMElement $element): DOMNode|bool|DOMElement|null { /* Element-Specific */ switch ( $element->tagName ) { case 'a': $element = $this->_parseAElement( $element ); break; case 'img': $element = $this->_parseImgElement( $element ); break; case 'iframe': $element = $this->_parseIframeElement( $element ); break; case 'video': $element = $this->_parseVideoElement( $element ); break; case 'audio': $element = $this->_parseAudioElement( $element ); break; } if ( !( $element instanceof DOMElement ) ) { return $element; } /* Anything which has a URL may need swapping out */ foreach ( array( 'href', 'src', 'srcset', 'data-ipshover-target', 'data-fileid', 'cite', 'action', 'longdesc', 'usemap', 'poster' ) as $attribute ) { if ( $element->hasAttribute( $attribute ) ) { if ( preg_match( '#^(https?:)?//(' . preg_quote( rtrim( str_replace( array( 'http://', 'https://' ), '', Settings::i()->base_url ), '/' ), '#' ) . ')/(.+?)$#', $element->getAttribute( $attribute ), $matches ) ) { $element->setAttribute( $attribute, '%7B___base_url___%7D/' . $matches[3] ); } } } foreach ( array( 'srcset', 'style' ) as $attribute ) { if ( $element->hasAttribute( $attribute ) ) { if ( mb_strpos( $element->getAttribute( $attribute ), Settings::i()->base_url ) ) { $element->setAttribute( $attribute, str_replace( Settings::i()->base_url, '%7B___base_url___%7D/', $element->getAttribute( $attribute ) ) ); } } } /* Return */ return $element; } /** * Parse element * * @param DOMElement $element The element * @return DOMElement */ protected function _parseAElement( DOMElement $element ) : DOMElement { /* Punycode it if necessary */ if ( !preg_match( '/^[\x00-\x7F]*$/', $element->getAttribute('href') ) ) { try { $punycodeEncoded = (string) Url::createFromString( $element->getAttribute('href') ); $element->setAttribute( 'href', $punycodeEncoded ); } catch(Url\Exception $e ) { } } /* If it's not allowed, remove the href */ if ( !static::isAllowedUrl( $element->getAttribute('href') ) ) { $element->removeAttribute( 'href' ); $element->setAttribute( 'class', 'ipsType_noLinkStyling' ); return $element; } /* Attachment? */ if ( $attachment = static::_getAttachment( $element->getAttribute('href'), $element->hasAttribute('data-fileid') ? $element->getAttribute('data-fileid') : NULL ) ) { $element->setAttribute( 'data-fileid', $attachment['attach_id'] ); $element->setAttribute( 'href', str_replace( array( 'http:', 'https:' ), '', str_replace( static::$fileObjectClasses['core_Attachment']->baseUrl(), '{fileStore.core_Attachment}', $element->getAttribute('href') ) ) ); $element->setAttribute( 'data-fileext', $attachment['attach_ext'] ); if ( ! empty( $attachment['attach_labels'] ) and $labels = static::getAttachmentLabels( $attachment ) ) { if ( count( $labels ) ) { if ( $this->_altLabelWord === NULL ) { $this->_altLabelWord = Lang::load( Lang::defaultLanguage() )->get( 'alt_label_could_be' ); } $element->setAttribute( 'alt', $this->_altLabelWord . ' ' . implode( ', ', $labels ) ); } } $this->_logAttachment( $attachment ); } /* Some other media? */ elseif ( $element->getAttribute('data-extension') and $file = $this->_getFile( $element->getAttribute('data-extension'), $element->getAttribute('href') ) ) { $element->setAttribute( 'href', '{fileStore.' . $file->storageExtension . '}/' . $file ); } try { $rels = $this->_getRelAttributes( Url::createFromString( $element->getAttribute('href') ) ); } catch(Url\Exception $e ) { $rels = array(); } /* Add rels */ $element->setAttribute( 'rel', implode( ' ', $rels ) ); return $element; } /** * @brief Emoticon Count */ protected int $_emoticons = 0; /** * Parse element * * @param DOMElement $element The element * @return bool|DOMElement */ protected function _parseImgElement( DOMElement $element ): bool|DOMElement { /* When editing content in the AdminCP, images and iframes get the src munged. When we save, we need to put that back */ $this->_removeMunge( $element ); if ( $element->getAttribute( 'class' ) and preg_match( '#ipsEmbedded_og__(favicon|image)#', $element->getAttribute( 'class' ) ) ) { return $element; } /* Is it an emoji? */ if ( $element->getAttribute('class') and in_array( 'ipsEmoji', explode( ' ', $element->getAttribute('class') ) ) and $element->getAttribute('alt') ) { $newElement = $element->ownerDocument->importNode( new DOMElement( 'span' ) ); $newElement->setAttribute( 'class', 'ipsEmoji' ); $newElement->appendChild( new DOMText( $element->getAttribute('alt') ) ); return $newElement; } /* If it's not allowed, remove the src */ try { static::isAllowedContentUrl( $element->getAttribute( 'src' ) ); } catch( UnexpectedValueException $e ) { $newElement = $element->ownerDocument->importNode( new DOMElement( 'span' ) ); $newElement->appendChild( new DOMText( $element->getAttribute( 'src' ) ) ); return $newElement; } /* Is it an emoticon? */ if ( $element->hasAttribute('data-emoticon') ) { if ( $this->_emoticons < 75 ) { if ( !isset( static::$fileObjectClasses['core_Emoticons'] ) ) { static::$fileObjectClasses['core_Emoticons'] = File::getClass('core_Emoticons' ); } $element->setAttribute( 'src', str_replace( array( 'http:', 'https:' ), '', str_replace( static::$fileObjectClasses['core_Emoticons']->baseUrl(), '{fileStore.core_Emoticons}', $element->getAttribute('src') ) ) ); if ( $srcSet = $element->getAttribute('srcset') ) { $element->setAttribute( 'srcset', str_replace( array( 'http:', 'https:' ), '', str_replace( static::$fileObjectClasses['core_Emoticons']->baseUrl(), '%7BfileStore.core_Emoticons%7D', $srcSet ) ) ); } $this->_emoticons++; } else { /* Set an attribute on the element - we'll need to know this later */ $element->setAttribute( 'data-ipsEmoticon-plain', $element->getAttribute('title') ); } } /* Or an attachment? */ elseif ( $attachment = static::_getAttachment( $element->getAttribute('src'), $element->hasAttribute('data-fileid') ? $element->getAttribute('data-fileid') : NULL ) ) { $file = $this->_getFile( 'core_Attachment', $element->getAttribute('src') ); $element->setAttribute( 'data-fileid', $attachment['attach_id'] ); $element->setAttribute( 'src', str_replace( array( 'http:', 'https:' ), '', str_replace( static::$fileObjectClasses['core_Attachment']->baseUrl(), '{fileStore.core_Attachment}', $element->getAttribute('src') ) ) ); if ( ( !$element->hasAttribute('alt') or $element->getAttribute('alt') == $file->filename ) and ! empty( $attachment['attach_labels'] ) and $labels = static::getAttachmentLabels( $attachment ) ) { if ( count( $labels ) ) { if ( $this->_altLabelWord === NULL ) { $this->_altLabelWord = Lang::load( Lang::defaultLanguage() )->get( 'alt_label_could_be' ); } $element->setAttribute( 'alt', $this->_altLabelWord . ' ' . implode( ', ', $labels ) ); } } if ( !$element->getAttribute('alt') ) { $element->setAttribute( 'alt', $attachment['attach_file'] ); } $this->_logAttachment( $attachment ); } /* Or some other media? */ elseif ( $element->getAttribute('data-extension') and $file = $this->_getFile( $element->getAttribute('data-extension'), $element->getAttribute('src') ) ) { $element->setAttribute( 'src', '{fileStore.' . $file->storageExtension . '}/' . $file ); if ( !$element->getAttribute('alt') ) { $element->setAttribute( 'alt', $file->originalFilename ); } } /* Nope, regular image */ else { /* We need an alt (HTMLPurifier handles this normally, but it may not always run) */ if ( !$element->getAttribute('alt') ) { $element->setAttribute( 'alt', mb_substr( basename( $element->getAttribute('src') ), 0, 40 ) ); } } /* Native Lazyload all imgs */ $element->setAttribute( 'loading', 'lazy' ); return $element; } /** * Parse