Файловый менеджер - Редактировать - /home/digilove/public_html/administrator/components/com_jmap/framework/aigenerator/readability.php
Назад
<?php /** * * @package JMAP::FRAMEWORK::administrator::components::com_jmap * @subpackage framework * @subpackage aigenerator * @author Joomla! Extensions Store * @copyright (C) 2021 - Joomla! Extensions Store * @license GNU/GPLv2 http://www.gnu.org/licenses/gpl-2.0.html */ // no direct access defined ( '_JEXEC' ) or die ( 'Restricted access' ); class JMapAigeneratorReadability { public $version = '1.7.1-without-multi-page'; public $convertLinksToFootnotes = false; public $revertForcedParagraphElements = true; public $articleTitle; public $articleContent; public $dom; public $url = null; // optional - URL where HTML was retrieved public $debug = false; public $lightClean = true; // preserves more content (experimental) added 2012-09-19 protected $body = null; // protected $bodyCache = null; // Cache the body HTML in case we need to re-use it later protected $flags = 7; // 1 | 2 | 4; // Start with all flags set. protected $success = false; // indicates whether we were able to extract or not protected $logger; /** * All of the regular expressions in use within readability. * Defined up here so we don't instantiate them repeatedly in loops. **/ public $regexps = array( 'unlikelyCandidates' => '/combx|comment|community|disqus|extra|foot|header|menu|remark|rss|shoutbox|sidebar|sponsor|ad\-break|agegate|pagination|pager|skip\-to\-text\-link|popup|flash|js|el__featured\-video|kicker|meta|kicker\-label|headline|page\-title/i', 'okMaybeItsACandidate' => '/and|article|body|column|main|continues|postContent|content|post|story|related|shadow|story\-content|story\-body\-supplemental|story\-body|story\-body\-text|story\-continues|story\-content|story\-body\-text|story\-continues\-2|el__leafmedia\-\-speakable\-paragraph|no\-js|story\-content|morning\-briefing\-weather\-module/i', 'positive' => '/article|body|story\-content|content|entry|hentry|main|page|attachment|pagination|post|text|blog|postContent|story|story\-body|story\-body\-supplemental|story\-body\-text|story\-continues|story\-content|story\-body\-text|story\-continues\-2|speakable|el__leafmedia\-\-speakable\-paragraph|articleBody|story\-content|story\-body|story\-body\-supplemental/i', 'negative' => '/combx|comment|skip\-to\-text\-link|com\-|contact|foot|footer|_nav|footnote|masthead|media|meta|outbrain|taboola|promo|related|scroll|shoutbox|sidebar|sponsor|shopping|tags|tool|widget|el__featured\-video|flash\-state|video__end\-slate\-\-inactive|js|zn\-large\-media|script|fav|unmute|el__video__collection__close\-\-expandable|js__video__collection__close\-\-expandable|kicker|meta|kicker\-label|headline|page\-title/i', 'divToPElements' => '/<(a|blockquote|dl|div|img|ol|p|pre|table|ul)/i', 'replaceBrs' => '/(<br[^>]*>[ \n\r\t]*){2,}/i', 'replaceFonts' => '/<(\/?)font[^>]*>/i', // 'trimRe' => '/^\s+|\s+$/g', // PHP has trim() 'normalize' => '/\s{2,}/u', 'killBreaks' => '/(<br\s*\/?>(\s| ?)*){1,}/', 'video' => '!//(player\.|www\.)?(youtube|vimeo|viddler)\.com!i', 'skipFootnoteLink' => '/^\s*(\[?[a-z0-9]{1,2}\]?|^|edit|citation needed)\s*$/i', ); /* constants */ const FLAG_STRIP_UNLIKELYS = 1; const FLAG_WEIGHT_CLASSES = 2; const FLAG_CLEAN_CONDITIONALLY = 4; /** * Create instance of Readability * * @param string UTF-8 encoded string * @param string (optional) URL associated with HTML (used for footnotes) * @param string which parser to use for turning raw HTML into a DOMDocument (either 'libxml' or 'html5lib') */ function __construct( $html, $url = null, $parser = 'libxml', $logging_function = false ) { $this->url = $url; $this->logger = $logging_function; /* Turn all double br's into p's */ $html = preg_replace( $this->regexps['replaceBrs'], '</p><p>', $html ); $html = preg_replace( $this->regexps['replaceFonts'], '<$1span>', $html ); $html = mb_encode_numericentity($html, [0x80, 0xFFFF, 0, 0xFFFF], 'UTF-8'); if ( trim( $html ) == '' ) { $html = '<html></html>'; } if ( $parser == 'html5lib' && ($this->dom = HTML5_Parser::parse( $html )) ) { // all good } else { $this->dom = new DOMDocument('1.0', 'UTF-8'); $this->dom->preserveWhiteSpace = false; @$this->dom->loadHTML( $html ); } $this->dom->registerNodeClass( 'DOMElement', 'JMapAigeneratorJslikehtmlelement' ); } /** * Get article title element * * @return DOMElement */ public function getTitle() { return $this->articleTitle; } /** * Get article content element * * @return DOMElement */ public function getContent() { return $this->articleContent; } /** * Runs readability. * * Workflow: * 1. Prep the document by removing script tags, css, etc. * 2. Build readability's DOM tree. * 3. Grab the article content from the current dom tree. * 4. Replace the current DOM tree with the new one. * 5. Read peacefully. * * @return boolean true if we found content, false otherwise **/ public function init() { if ( ! isset( $this->dom->documentElement ) ) { return false; } $this->removeScripts( $this->dom ); // die($this->getInnerHTML($this->dom->documentElement)); // Assume successful outcome $this->success = true; $bodyElems = $this->dom->getElementsByTagName( 'body' ); if ( $bodyElems->length > 0 ) { if ( $this->bodyCache == null ) { $this->bodyCache = $bodyElems->item( 0 )->innerHTML; } if ( $this->body == null ) { $this->body = $bodyElems->item( 0 ); } } $this->prepDocument(); // die($this->dom->documentElement->parentNode->nodeType); // $this->setInnerHTML($this->dom->documentElement, $this->getInnerHTML($this->dom->documentElement)); // die($this->getInnerHTML($this->dom->documentElement)); /* Build readability's DOM tree */ $overlay = $this->dom->createElement( 'div' ); $innerDiv = $this->dom->createElement( 'div' ); $articleTitle = $this->getArticleTitle(); $articleContent = $this->grabArticle(); if ( ! $articleContent ) { $this->success = false; $articleContent = $this->dom->createElement( 'div' ); $articleContent->setAttribute( 'id', 'readability-content' ); $articleContent->innerHTML = '<p>Sorry, Readability was unable to parse this page for content.</p>'; } $overlay->setAttribute( 'id', 'readOverlay' ); $innerDiv->setAttribute( 'id', 'readInner' ); /* Glue the structure of our document together. */ $innerDiv->appendChild( $articleTitle ); $innerDiv->appendChild( $articleContent ); $overlay->appendChild( $innerDiv ); /* Clear the old HTML, insert the new content. */ $this->body->innerHTML = ''; $this->body->appendChild( $overlay ); // document.body.insertBefore(overlay, document.body.firstChild); $this->body->removeAttribute( 'style' ); $this->postProcessContent( $articleContent ); // Set title and content instance variables $this->articleTitle = $articleTitle; $this->articleContent = $articleContent; return $this->success; } protected function external_logger( $msg ) { if ( false !== $this->logger && is_callable( $this->logger ) ) { call_user_func_array( $this->logger, array( $msg ) ); } } /** * Check if the url is an absolute URL in some way * * @param string $url * @return bool */ protected function isFullyQualified($url) { $isFullyQualified = substr($url, 0, 7) == 'http://' || substr($url, 0, 8) == 'https://' || substr($url, 0, 2) == '//'; return $isFullyQualified; } /** * Get the top level host domain for each kind of URL needed to avoid redirects on CURL exec * * @access private * @param string $url * @return string */ protected function getHost($url) { if (strpos ( $url, "http" ) !== false) { $httpurl = $url; } else { $httpurl = "https://" . $url; } $parse = parse_url ( $httpurl ); $scheme = $parse ['scheme']; $domain = $parse ['host']; return $scheme . '://' . $domain; } /** * Debug */ protected function dbg( $msg ) { if ( $this->debug ) { echo '* ',$msg, "\n"; } $this->external_logger( $msg ); } /** * Run any post-process modifications to article content as necessary. * * @param DOMElement * @return void */ public function postProcessContent( $articleContent ) { if ( $this->convertLinksToFootnotes && ! preg_match( '/wikipedia\.org/', @$this->url ) ) { $this->addFootnotes( $articleContent ); } } /** * Get the article title as an H1. * * @return DOMElement */ protected function getArticleTitle() { $curTitle = ''; $origTitle = ''; try { $curTitle = $origTitle = $this->getInnerText( $this->dom->getElementsByTagName( 'title' )->item( 0 ) ); } catch (Exception $e) {} if ( preg_match( '/ [\|\-] /', $curTitle ) ) { $curTitle = preg_replace( '/(.*)[\|\-] .*/i', '$1', $origTitle ); if ( count( explode( ' ', $curTitle ) ) < 3 ) { $curTitle = preg_replace( '/[^\|\-]*[\|\-](.*)/i', '$1', $origTitle ); } } elseif ( strpos( $curTitle, ': ' ) !== false ) { $curTitle = preg_replace( '/.*:(.*)/i', '$1', $origTitle ); if ( count( explode( ' ', $curTitle ) ) < 3 ) { $curTitle = preg_replace( '/[^:]*[:](.*)/i','$1', $origTitle ); } } elseif ( strlen( $curTitle ) > 150 || strlen( $curTitle ) < 15 ) { $hOnes = $this->dom->getElementsByTagName( 'h1' ); if ( $hOnes->length == 1 ) { $curTitle = $this->getInnerText( $hOnes->item( 0 ) ); } } $curTitle = trim( $curTitle ); if ( count( explode( ' ', $curTitle ) ) <= 4 ) { $curTitle = $origTitle; } $articleTitle = $this->dom->createElement( 'h1' ); $articleTitle->innerHTML = $curTitle; return $articleTitle; } /** * Prepare the HTML document for readability to scrape it. * This includes things like stripping javascript, CSS, and handling terrible markup. * * @return void **/ protected function prepDocument() { /** * In some cases a body element can't be found (if the HTML is totally hosed for example) * so we create a new body node and append it to the document. */ if ( $this->body == null ) { $this->body = $this->dom->createElement( 'body' ); $this->dom->documentElement->appendChild( $this->body ); } $this->body->setAttribute( 'id', 'readabilityBody' ); /* Remove all style tags in head */ $styleTags = $this->dom->getElementsByTagName( 'style' ); for ( $i = $styleTags->length -1; $i >= 0; $i-- ) { $styleTags->item( $i )->parentNode->removeChild( $styleTags->item( $i ) ); } /* Turn all double br's into p's */ /* Note, this is pretty costly as far as processing goes. Maybe optimize later. */ // document.body.innerHTML = document.body.innerHTML.replace(readability.regexps.replaceBrs, '</p><p>').replace(readability.regexps.replaceFonts, '<$1span>'); // We do this in the constructor for PHP as that's when we have raw HTML - before parsing it into a DOM tree. // Manipulating innerHTML as it's done in JS is not possible in PHP. } /** * For easier reading, convert this document to have footnotes at the bottom rather than inline links. * * @see http://www.roughtype.com/archives/2010/05/experiments_in.php * * @return void **/ public function addFootnotes( $articleContent ) { $footnotesWrapper = $this->dom->createElement( 'div' ); $footnotesWrapper->setAttribute( 'id', 'readability-footnotes' ); $footnotesWrapper->innerHTML = '<h3>References</h3>'; $articleFootnotes = $this->dom->createElement( 'ol' ); $articleFootnotes->setAttribute( 'id', 'readability-footnotes-list' ); $footnotesWrapper->appendChild( $articleFootnotes ); $articleLinks = $articleContent->getElementsByTagName( 'a' ); $linkCount = 0; for ( $i = 0; $i < $articleLinks->length; $i++ ) { $articleLink = $articleLinks->item( $i ); $footnoteLink = $articleLink->cloneNode( true ); $refLink = $this->dom->createElement( 'a' ); $footnote = $this->dom->createElement( 'li' ); $linkDomain = @parse_url( $footnoteLink->getAttribute( 'href' ), PHP_URL_HOST ); if ( ! $linkDomain && isset( $this->url ) ) { $linkDomain = @parse_url( $this->url, PHP_URL_HOST ); } // linkDomain = footnoteLink.host ? footnoteLink.host : document.location.host, $linkText = $this->getInnerText( $articleLink ); if ( (strpos( $articleLink->getAttribute( 'class' ), 'readability-DoNotFootnote' ) !== false) || preg_match( $this->regexps['skipFootnoteLink'], $linkText ) ) { continue; } $linkCount++; /** Add a superscript reference after the article link */ $refLink->setAttribute( 'href', '#readabilityFootnoteLink-' . $linkCount ); $refLink->innerHTML = '<small><sup>[' . $linkCount . ']</sup></small>'; $refLink->setAttribute( 'class', 'readability-DoNotFootnote' ); $refLink->setAttribute( 'style', 'color: inherit;' ); // does this work or should we use DOMNode.isSameNode()? if ( $articleLink->parentNode->lastChild == $articleLink ) { $articleLink->parentNode->appendChild( $refLink ); } else { $articleLink->parentNode->insertBefore( $refLink, $articleLink->nextSibling ); } $articleLink->setAttribute( 'style', 'color: inherit; text-decoration: none;' ); $articleLink->setAttribute( 'name', 'readabilityLink-' . $linkCount ); $footnote->innerHTML = '<small><sup><a href="#readabilityLink-' . $linkCount . '" title="Jump to Link in Article">^</a></sup></small> '; $footnoteLink->innerHTML = ($footnoteLink->getAttribute( 'title' ) != '' ? $footnoteLink->getAttribute( 'title' ) : $linkText); $footnoteLink->setAttribute( 'name', 'readabilityFootnoteLink-' . $linkCount ); $footnote->appendChild( $footnoteLink ); if ( $linkDomain ) { $footnote->innerHTML = $footnote->innerHTML . '<small> (' . $linkDomain . ')</small>'; } $articleFootnotes->appendChild( $footnote ); } if ( $linkCount > 0 ) { $articleContent->appendChild( $footnotesWrapper ); } } /** * Reverts P elements with class 'readability-styled' * to text nodes - which is what they were before. * * @param DOMElement * @return void */ function revertReadabilityStyledElements( $articleContent ) { $xpath = new \DOMXPath( $articleContent->ownerDocument ); $elems = $xpath->query( './/p[@class="readability-styled"]', $articleContent ); // $elems = $articleContent->getElementsByTagName('p'); for ( $i = $elems->length -1; $i >= 0; $i-- ) { $e = $elems->item( $i ); $e->parentNode->replaceChild( $articleContent->ownerDocument->createTextNode( $e->textContent ), $e ); // if ($e->hasAttribute('class') && $e->getAttribute('class') == 'readability-styled') { // $e->parentNode->replaceChild($this->dom->createTextNode($e->textContent), $e); // } } } /** * Prepare the article node for display. Clean out any inline styles, * iframes, forms, strip extraneous <p> tags, etc. * * @param DOMElement * @return void */ function prepArticle( $articleContent ) { $this->cleanStyles( $articleContent ); $this->killBreaks( $articleContent ); if ( $this->revertForcedParagraphElements ) { $this->revertReadabilityStyledElements( $articleContent ); } /* Clean out junk from the article content */ $this->cleanConditionally( $articleContent, 'form' ); $this->clean( $articleContent, 'object' ); $this->clean( $articleContent, 'script' ); $this->clean( $articleContent, 'link' ); $this->clean( $articleContent, 'meta' ); // $this->clean($articleContent, 'h1'); /** * If there is only one h2, they are probably using it * as a header and not a subheader, so remove it since we already have a header. */ if ( ! $this->lightClean && ($articleContent->getElementsByTagName( 'h2' )->length == 1) ) { // $this->clean($articleContent, 'h2'); } $this->clean( $articleContent, 'iframe' ); $this->cleanHeaders( $articleContent ); /* Do these last as the previous stuff may have removed junk that will affect these */ $this->cleanConditionally( $articleContent, 'table' ); $this->cleanConditionally( $articleContent, 'ul' ); $this->cleanConditionally( $articleContent, 'div' ); // Make imgs always absolute URLs, use 2 methods: getElementsByTagName and XPath $imagesFoundDOM = $articleContent->getElementsByTagName( 'img' ); if( $imagesFoundDOM->length > 0) { for ( $im = $imagesFoundDOM->length -1; $im >= 0; $im-- ) { $imgTag = $imagesFoundDOM->item( $im ); $originalSrc = $imgTag->getAttribute('src'); if($originalSrc && !$this->isFullyQualified($originalSrc) && !preg_match('/data:image/i', $originalSrc)) { // Relative image has been found, go on to append the original src domain $originalSrc = $this->getHost($this->url) . '/' . ltrim($originalSrc, '/'); $imgTag->setAttributeNode(new \DOMAttr('src', $originalSrc)); } // Support for custom lazy src attributes $arrayOfDataSrcAttributes = [ 'data-src', 'data-lazyload', 'data-original', 'data-dt-lazy-src', 'data-lazy-src' ]; foreach ($arrayOfDataSrcAttributes as $singleDataSrcAttribute) { if($imgTag->hasAttribute($singleDataSrcAttribute)) { $originalDataSrc = $imgTag->getAttribute($singleDataSrcAttribute); if($originalDataSrc) { if(!$this->isFullyQualified($originalDataSrc)) { $imgTag->setAttributeNode(new \DOMAttr($singleDataSrcAttribute, $this->getHost($this->url) . '/' . ltrim($originalDataSrc, '/'))); } // Nullify the src attribute for the regular expression in the purify contents $imgTag->setAttributeNode(new \DOMAttr('src', 'data:image/png;')); } } } } } /* Remove extra paragraphs */ $articleParagraphs = $articleContent->getElementsByTagName( 'p' ); for ( $i = $articleParagraphs->length -1; $i >= 0; $i-- ) { $imgCount = $articleParagraphs->item( $i )->getElementsByTagName( 'img' )->length; $embedCount = $articleParagraphs->item( $i )->getElementsByTagName( 'embed' )->length; $objectCount = $articleParagraphs->item( $i )->getElementsByTagName( 'object' )->length; $iframeCount = $articleParagraphs->item( $i )->getElementsByTagName( 'iframe' )->length; if ( $imgCount === 0 && $embedCount === 0 && $objectCount === 0 && $iframeCount === 0 && $this->getInnerText( $articleParagraphs->item( $i ), false ) == '' ) { $articleParagraphs->item( $i )->parentNode->removeChild( $articleParagraphs->item( $i ) ); } } try { $articleContent->innerHTML = preg_replace( '/<br[^>]*>\s*<p/i', '<p', $articleContent->innerHTML ); // articleContent.innerHTML = articleContent.innerHTML.replace(/<br[^>]*>\s*<p/gi, '<p'); } catch (Exception $e) { $this->dbg( 'Cleaning innerHTML of breaks failed. This is an IE strict-block-elements bug. Ignoring.: ' . $e ); } } /** * Initialize a node with the readability object. Also checks the * className/id for special names to add to its score. * * @param Element * @return void **/ protected function initializeNode( $node ) { $readability = $this->dom->createAttribute( 'readability' ); $readability->value = 0; // this is our contentScore $node->setAttributeNode( $readability ); switch ( strtoupper( $node->tagName ) ) { // unsure if strtoupper is needed, but using it just in case case 'DIV': $readability->value += 5; break; case 'PRE': case 'TD': case 'BLOCKQUOTE': $readability->value += 3; break; case 'ADDRESS': case 'OL': case 'UL': case 'DL': case 'DD': case 'DT': case 'LI': case 'FORM': $readability->value -= 3; break; case 'H1': case 'H2': case 'H3': case 'H4': case 'H5': case 'H6': case 'TH': $readability->value -= 5; break; case 'SCRIPT': $readability->value -= 100; break; } $readability->value += $this->getClassWeight( $node ); } /** * * grabArticle - Using a variety of metrics (content score, classname, element types), find the content that is * most likely to be the stuff a user wants to read. Then return it wrapped up in a div. * * @return DOMElement **/ protected function grabArticle( $page = null ) { $stripUnlikelyCandidates = $this->flagIsActive( self::FLAG_STRIP_UNLIKELYS ); if ( ! $page ) { $page = $this->dom; } $allElements = $page->getElementsByTagName( '*' ); /** * First, node prepping. Trash nodes that look cruddy (like ones with the class name "comment", etc), and turn divs * into P tags where they have been used inappropriately (as in, where they contain no other block level elements.) * * Note: Assignment from index for performance. See http://www.peachpit.com/articles/article.aspx?p=31567&seqNum=5 * Shouldn't this be a reverse traversal? */ $node = null; $nodesToScore = array(); //var_dump('<pre>',$allElements->item( $nodeIndex )); //die(); for ( $nodeIndex = 0; ($node = $allElements->item( $nodeIndex )); $nodeIndex++ ) { // for ($nodeIndex=$targetList->length-1; $nodeIndex >= 0; $nodeIndex--) { // $node = $targetList->item($nodeIndex); $tagName = strtoupper( $node->tagName ); //var_dump($node); /* Remove unlikely candidates */ if ( $stripUnlikelyCandidates ) { $unlikelyMatchString = $node->getAttribute( 'class' ) . $node->getAttribute( 'id' ); if ( preg_match( $this->regexps['unlikelyCandidates'], $unlikelyMatchString ) && ! preg_match( $this->regexps['okMaybeItsACandidate'], $unlikelyMatchString ) && $tagName != 'BODY' ) { $this->dbg( 'Removing unlikely candidate - ' . $unlikelyMatchString ); // $nodesToRemove[] = $node; $node->parentNode->removeChild( $node ); $nodeIndex--; continue; } } if ( $tagName == 'P' || $tagName == 'TD' || $tagName == 'PRE' ) { $nodesToScore[] = $node; } /* Turn all divs that don't have children block level elements into p's */ if ( $tagName == 'DIV' ) { if ( ! preg_match( $this->regexps['divToPElements'], $node->innerHTML ) ) { // $this->dbg('Altering div to p'); $newNode = $this->dom->createElement( 'p' ); try { $newNode->innerHTML = $node->innerHTML; // $nodesToReplace[] = array('new'=>$newNode, 'old'=>$node); $node->parentNode->replaceChild( $newNode, $node ); $nodeIndex--; $nodesToScore[] = $node; // or $newNode? } catch (Exception $e) { $this->dbg( 'Could not alter div to p, reverting back to div.: ' . $e ); } } else { /* EXPERIMENTAL */ // change these p elements back to text nodes after processing for ( $i = 0, $il = $node->childNodes->length; $i < $il; $i++ ) { $childNode = $node->childNodes->item( $i ); if ( $childNode->nodeType == 3 ) { // XML_TEXT_NODE // $this->dbg('replacing text node with a p tag with the same content.'); $p = $this->dom->createElement( 'p' ); $p->innerHTML = $childNode->nodeValue; $p->setAttribute( 'style', 'display: inline;' ); $p->setAttribute( 'class', 'readability-styled' ); $childNode->parentNode->replaceChild( $p, $childNode ); } } } } } /** * Loop through all paragraphs, and assign a score to them based on how content-y they look. * Then add their score to their parent node. * * A score is determined by things like number of commas, class names, etc. Maybe eventually link density. */ $candidates = array(); for ( $pt = 0; $pt < count( $nodesToScore ); $pt++ ) { $parentNode = $nodesToScore[ $pt ]->parentNode; // $grandParentNode = $parentNode ? $parentNode->parentNode : null; $grandParentNode = ! $parentNode ? null : (($parentNode->parentNode instanceof DOMElement) ? $parentNode->parentNode : null); $grandGrandParentNode = ! $grandParentNode ? null : (($grandParentNode->parentNode instanceof DOMElement) ? $grandParentNode->parentNode : null); $innerText = $this->getInnerText( $nodesToScore[ $pt ] ); if ( ! $parentNode || ! isset( $parentNode->tagName ) ) { continue; } /* If this paragraph is less than 25 characters, don't even count it. */ if ( strlen( $innerText ) < 25 ) { continue; } /* Initialize readability data for the parent. */ if ( ! $parentNode->hasAttribute( 'readability' ) ) { $this->initializeNode( $parentNode ); $candidates[] = $parentNode; } /* Initialize readability data for the grandparent. */ if ( $grandParentNode && ! $grandParentNode->hasAttribute( 'readability' ) && isset( $grandParentNode->tagName ) ) { $this->initializeNode( $grandParentNode ); $candidates[] = $grandParentNode; } /* Initialize readability data for the grandgrandparent. */ if ( $grandGrandParentNode && ! $grandGrandParentNode->hasAttribute( 'readability' ) && isset( $grandGrandParentNode->tagName ) ) { $this->initializeNode( $grandGrandParentNode ); $candidates[] = $grandGrandParentNode; } $contentScore = 0; /* Add a point for the paragraph itself as a base. */ $contentScore++; /* Add points for any commas within this paragraph */ $contentScore += count( explode( ',', $innerText ) ); /* For every 100 characters in this paragraph, add another point. Up to 3 points. */ $contentScore += min( floor( strlen( $innerText ) / 100 ), 3 ); /* Add the score to the parent. The grandparent gets half. */ $parentNode->getAttributeNode( 'readability' )->value += $contentScore; if ( $grandParentNode ) { $grandParentNode->getAttributeNode( 'readability' )->value += $contentScore / 2; } } /** * After we've calculated scores, loop through all of the possible candidate nodes we found * and find the one with the highest score. */ $topCandidate = null; for ( $c = 0, $cl = count( $candidates ); $c < $cl; $c++ ) { /** * Scale the final candidates score based on link density. Good content should have a * relatively small link density (5% or less) and be mostly unaffected by this operation. */ $readability = $candidates[ $c ]->getAttributeNode( 'readability' ); $readability->value = $readability->value * (1 -$this->getLinkDensity( $candidates[ $c ] )); if ( 'article' == $candidates[ $c ]->tagName ) { $readability->value = $readability->value + 300; // var_dump($candidates[$c]); die(); } $this->dbg( 'Candidate: ' . $candidates[ $c ]->tagName . ' (' . $candidates[ $c ]->getAttribute( 'class' ) . ':' . $candidates[ $c ]->getAttribute( 'id' ) . ') with score ' . $readability->value ); if ( ! $topCandidate || $readability->value > (int) $topCandidate->getAttribute( 'readability' ) ) { $topCandidate = $candidates[ $c ]; } } /** * If we still have no top candidate, just use the body as a last resort. * We also have to copy the body node so it is something we can modify. */ if ( $topCandidate === null || strtoupper( $topCandidate->tagName ) == 'BODY' ) { $topCandidate = $this->dom->createElement( 'div' ); if ( $page instanceof DOMDocument ) { if ( ! isset( $page->documentElement ) ) { // we don't have a body either? what a mess! :) } else { $topCandidate->innerHTML = $page->documentElement->innerHTML; $page->documentElement->innerHTML = ''; $page->documentElement->appendChild( $topCandidate ); } } else { $topCandidate->innerHTML = $page->innerHTML; $page->innerHTML = ''; $page->appendChild( $topCandidate ); } $this->initializeNode( $topCandidate ); } /** * Now that we have the top candidate, look through its siblings for content that might also be related. * Things like preambles, content split by ads that we removed, etc. */ $articleContent = $this->dom->createElement( 'div' ); $articleContent->setAttribute( 'id', 'readability-content' ); $siblingScoreThreshold = max( 10, ((int) $topCandidate->getAttribute( 'readability' )) * 0.2 ); $siblingNodes = $topCandidate->parentNode->childNodes; if ( ! isset( $siblingNodes ) ) { $siblingNodes = new stdClass; $siblingNodes->length = 0; } for ( $s = 0, $sl = $siblingNodes->length; $s < $sl; $s++ ) { $siblingNode = $siblingNodes->item( $s ); $append = false; $this->dbg( 'Looking at sibling node: ' . $siblingNode->nodeName . (($siblingNode->nodeType === XML_ELEMENT_NODE && $siblingNode->hasAttribute( 'readability' )) ? (' with score ' . $siblingNode->getAttribute( 'readability' )) : '') ); // dbg('Sibling has score ' . ($siblingNode->readability ? siblingNode.readability.contentScore : 'Unknown')); if ( $siblingNode === $topCandidate ) { $append = true; } $contentBonus = 0; /* Give a bonus if sibling nodes and top candidates have the example same classname */ if ( $siblingNode->nodeType === XML_ELEMENT_NODE && $siblingNode->getAttribute( 'class' ) == $topCandidate->getAttribute( 'class' ) && $topCandidate->getAttribute( 'class' ) != '' ) { $contentBonus += ((int) $topCandidate->getAttribute( 'readability' )) * 0.2; } if ( $siblingNode->nodeType === XML_ELEMENT_NODE && $siblingNode->hasAttribute( 'readability' ) && (((int) $siblingNode->getAttribute( 'readability' )) + $contentBonus) >= $siblingScoreThreshold ) { $append = true; } if ( strtoupper( $siblingNode->nodeName ) == 'P' ) { $linkDensity = $this->getLinkDensity( $siblingNode ); $nodeContent = $this->getInnerText( $siblingNode ); $nodeLength = strlen( $nodeContent ); if ( $nodeLength > 80 && $linkDensity < 0.25 ) { $append = true; } elseif ( $nodeLength < 80 && $linkDensity === 0 && preg_match( '/\.( |$)/', $nodeContent ) ) { $append = true; } } if ( $append ) { $this->dbg( 'Appending node: ' . $siblingNode->nodeName ); $nodeToAppend = null; $sibNodeName = strtoupper( $siblingNode->nodeName ); if ( $sibNodeName != 'DIV' && $sibNodeName != 'P' ) { /* We have a node that isn't a common block level element, like a form or td tag. Turn it into a div so it doesn't get filtered out later by accident. */ $this->dbg( 'Altering siblingNode of ' . $sibNodeName . ' to div.' ); $nodeToAppend = $this->dom->createElement( 'div' ); try { $nodeToAppend->setAttribute( 'id', $siblingNode->getAttribute( 'id' ) ); $nodeToAppend->innerHTML = $siblingNode->innerHTML; } catch (Exception $e) { $this->dbg( 'Could not alter siblingNode to div, reverting back to original.' ); $nodeToAppend = $siblingNode; $s--; $sl--; } } else { $nodeToAppend = $siblingNode; $s--; $sl--; } /* To ensure a node does not interfere with readability styles, remove its classnames */ $nodeToAppend->removeAttribute( 'class' ); /* Append sibling and subtract from our list because it removes the node when you append to another node */ $articleContent->appendChild( $nodeToAppend ); } } /** * So we have all of the content that we need. Now we clean it up for presentation. */ $this->prepArticle( $articleContent ); /** * Now that we've gone through the full algorithm, check to see if we got any meaningful content. * If we didn't, we may need to re-run grabArticle with different flags set. This gives us a higher * likelihood of finding the content, and the sieve approach gives us a higher likelihood of * finding the -right- content. */ if ( strlen( $this->getInnerText( $articleContent, false ) ) < 250 ) { // find out why element disappears sometimes, e.g. for this URL http://www.businessinsider.com/6-hedge-fund-etfs-for-average-investors-2011-7 // in the meantime, we check and create an empty element if it's not there. if ( ! isset( $this->body->childNodes ) ) { $this->body = $this->dom->createElement( 'body' ); } $this->body->innerHTML = $this->bodyCache; if ( $this->flagIsActive( self::FLAG_STRIP_UNLIKELYS ) ) { $this->removeFlag( self::FLAG_STRIP_UNLIKELYS ); return $this->grabArticle( $this->body ); } elseif ( $this->flagIsActive( self::FLAG_WEIGHT_CLASSES ) ) { $this->removeFlag( self::FLAG_WEIGHT_CLASSES ); return $this->grabArticle( $this->body ); } elseif ( $this->flagIsActive( self::FLAG_CLEAN_CONDITIONALLY ) ) { $this->removeFlag( self::FLAG_CLEAN_CONDITIONALLY ); return $this->grabArticle( $this->body ); } else { return false; } } return $articleContent; } /** * Remove script tags from document * * @param DOMElement * @return void */ public function removeScripts( $doc ) { $scripts = $doc->getElementsByTagName( 'script' ); for ( $i = $scripts->length -1; $i >= 0; $i-- ) { $scripts->item( $i )->parentNode->removeChild( $scripts->item( $i ) ); } } /** * Get the inner text of a node. * This also strips out any excess whitespace to be found. * * @param DOMElement $ * @param boolean $normalizeSpaces (default: true) * @return string **/ public function getInnerText( $e, $normalizeSpaces = true ) { $textContent = ''; if ( ! isset( $e->textContent ) || $e->textContent == '' ) { return ''; } $textContent = trim( $e->textContent ); if ( $normalizeSpaces ) { return preg_replace( $this->regexps['normalize'], ' ', $textContent ); } else { return $textContent; } } /** * Get the number of times a string $s appears in the node $e. * * @param DOMElement $e * @param string - what to count. Default is "," * @return number (integer) **/ public function getCharCount( $e, $s = ',' ) { return substr_count( $this->getInnerText( $e ), $s ); } /** * Remove the style attribute on every $e and under. * * @param DOMElement $e * @return void */ public function cleanStyles( $e ) { if ( ! is_object( $e ) ) { return; } $elems = $e->getElementsByTagName( '*' ); foreach ( $elems as $elem ) { $elem->removeAttribute( 'style' ); } } /** * Get the density of links as a percentage of the content * This is the amount of text that is inside a link divided by the total text in the node. * * @param DOMElement $e * @return number (float) */ public function getLinkDensity( $e ) { $links = $e->getElementsByTagName( 'a' ); $textLength = strlen( $this->getInnerText( $e ) ); $linkLength = 0; for ( $i = 0, $il = $links->length; $i < $il; $i++ ) { $linkLength += strlen( $this->getInnerText( $links->item( $i ) ) ); } if ( $textLength > 0 ) { return $linkLength / $textLength; } else { return 0; } } /** * Get an elements class/id weight. Uses regular expressions to tell if this * element looks good or bad. * * @param DOMElement $e * @return number (Integer) */ public function getClassWeight( $e ) { if ( ! $this->flagIsActive( self::FLAG_WEIGHT_CLASSES ) ) { return 0; } $weight = 0; /* Look for a special classname */ if ( $e->hasAttribute( 'class' ) && $e->getAttribute( 'class' ) != '' ) { if ( preg_match( $this->regexps['negative'], $e->getAttribute( 'class' ) ) ) { $weight -= 25; } if ( preg_match( $this->regexps['positive'], $e->getAttribute( 'class' ) ) ) { $weight += 25; } } //var_dump('<pre>', $e); //die(); /* Look for a special ID */ if ( $e->hasAttribute( 'id' ) && $e->getAttribute( 'id' ) != '' ) { if ( preg_match( $this->regexps['negative'], $e->getAttribute( 'id' ) ) ) { $weight -= 55; } if ( preg_match( $this->regexps['positive'], $e->getAttribute( 'id' ) ) ) { $weight += 55; } if ( 'story' == $e->getAttribute( 'id' ) && 'article' == $e->tagName ){ $weight += 300; } if ( 'main' == $e->getAttribute( 'id' ) && 'main' == $e->tagName ){ $weight += 300; } } if ( $e->hasAttribute( 'itemprop' ) && $e->getAttribute( 'itemprop' ) != '' ) { if ( preg_match( $this->regexps['negative'], $e->getAttribute( 'itemprop' ) ) ) { $weight -= 25; } if ( preg_match( $this->regexps['positive'], $e->getAttribute( 'itemprop' ) ) ) { $weight += 25; } if ( 'articleBody' == $e->getAttribute( 'itemprop' ) ) { $weight += 400; } } if ( $e->hasAttribute( 'role' ) && $e->getAttribute( 'role' ) != '' ) { if ( preg_match( $this->regexps['negative'], $e->getAttribute( 'role' ) ) ) { $weight -= 25; } if ( preg_match( $this->regexps['positive'], $e->getAttribute( 'role' ) ) ) { $weight += 25; } if ( 'main' == $e->getAttribute( 'role' ) ) { $weight += 400; } } if ( $e->hasAttribute( 'data-para-count' ) && $e->getAttribute( 'data-para-count' ) != '' ) { if ( intval($e->getAttribute( 'data-para-count')) > 0 ) { $weight += 200; } } return $weight; } /** * Remove extraneous break tags from a node. * * @param DOMElement $node * @return void */ public function killBreaks( $node ) { $html = $node->innerHTML; $html = preg_replace( $this->regexps['killBreaks'], '<br />', $html ); $node->innerHTML = $html; } /** * Clean a node of all elements of type "tag". * (Unless it's a youtube/vimeo video. People love movies.) * * Updated 2012-09-18 to preserve youtube/vimeo iframes * * @param DOMElement $e * @param string $tag * @return void */ public function clean( $e, $tag ) { $targetList = $e->getElementsByTagName( $tag ); $isEmbed = ($tag == 'iframe' || $tag == 'object' || $tag == 'embed'); for ( $y = $targetList->length -1; $y >= 0; $y-- ) { /* Allow youtube and vimeo videos through as people usually want to see those. */ if ( $isEmbed ) { $attributeValues = ''; for ( $i = 0, $il = $targetList->item( $y )->attributes->length; $i < $il; $i++ ) { $attributeValues .= $targetList->item( $y )->attributes->item( $i )->value . '|'; // DOMAttr } /* First, check the elements attributes to see if any of them contain youtube or vimeo */ if ( preg_match( $this->regexps['video'], $attributeValues ) ) { continue; } /* Then check the elements inside this element for the same. */ if ( preg_match( $this->regexps['video'], $targetList->item( $y )->innerHTML ) ) { continue; } } $targetList->item( $y )->parentNode->removeChild( $targetList->item( $y ) ); } } /** * Clean an element of all tags of type "tag" if they look fishy. * "Fishy" is an algorithm based on content length, classnames, * link density, number of images & embeds, etc. * * @param DOMElement $e * @param string $tag * @return void */ public function cleanConditionally( $e, $tag ) { if ( ! $this->flagIsActive( self::FLAG_CLEAN_CONDITIONALLY ) ) { return; } $tagsList = $e->getElementsByTagName( $tag ); $curTagsLength = $tagsList->length; /** * Gather counts for other typical elements embedded within. * Traverse backwards so we can remove nodes at the same time without effecting the traversal. * * Consider taking into account original contentScore here. */ for ( $i = $curTagsLength -1; $i >= 0; $i-- ) { $weight = $this->getClassWeight( $tagsList->item( $i ) ); $contentScore = ($tagsList->item( $i )->hasAttribute( 'readability' )) ? (int) $tagsList->item( $i )->getAttribute( 'readability' ) : 0; $this->dbg( 'Cleaning Conditionally ' . $tagsList->item( $i )->tagName . ' (' . $tagsList->item( $i )->getAttribute( 'class' ) . ':' . $tagsList->item( $i )->getAttribute( 'id' ) . ')' . (($tagsList->item( $i )->hasAttribute( 'readability' )) ? (' with score ' . $tagsList->item( $i )->getAttribute( 'readability' )) : '') ); if ( $weight + $contentScore < 0 ) { $tagsList->item( $i )->parentNode->removeChild( $tagsList->item( $i ) ); } elseif ( $this->getCharCount( $tagsList->item( $i ), ',' ) < 10 ) { /** * If there are not very many commas, and the number of * non-paragraph elements is more than paragraphs or other ominous signs, remove the element. */ $p = $tagsList->item( $i )->getElementsByTagName( 'p' )->length; $img = $tagsList->item( $i )->getElementsByTagName( 'img' )->length; $li = $tagsList->item( $i )->getElementsByTagName( 'li' )->length -100; $input = $tagsList->item( $i )->getElementsByTagName( 'input' )->length; $a = $tagsList->item( $i )->getElementsByTagName( 'a' )->length; $embedCount = 0; $embeds = $tagsList->item( $i )->getElementsByTagName( 'embed' ); for ( $ei = 0, $il = $embeds->length; $ei < $il; $ei++ ) { if ( preg_match( $this->regexps['video'], $embeds->item( $ei )->getAttribute( 'src' ) ) ) { $embedCount++; } } $embeds = $tagsList->item( $i )->getElementsByTagName( 'iframe' ); for ( $ei = 0, $il = $embeds->length; $ei < $il; $ei++ ) { if ( preg_match( $this->regexps['video'], $embeds->item( $ei )->getAttribute( 'src' ) ) ) { $embedCount++; } } $linkDensity = $this->getLinkDensity( $tagsList->item( $i ) ); $contentLength = strlen( $this->getInnerText( $tagsList->item( $i ) ) ); $toRemove = false; if ( $this->lightClean ) { $this->dbg( 'Light clean...' ); if ( ($img > $p) && ($img > 4) ) { $this->dbg( ' more than 4 images and more image elements than paragraph elements' ); $toRemove = true; } elseif ( $li > $p && $tag != 'ul' && $tag != 'ol' ) { $this->dbg( ' too many <li> elements, and parent is not <ul> or <ol>' ); $toRemove = true; } elseif ( $input > floor( $p / 3 ) ) { $this->dbg( ' too many <input> elements' ); $toRemove = true; } elseif ( $contentLength < 25 && ($embedCount === 0 && ($img === 0 || $img > 2)) ) { $this->dbg( ' content length less than 25 chars, 0 embeds and either 0 images or more than 2 images' ); $toRemove = true; } elseif ( $weight < 25 && $linkDensity > 0.2 ) { $this->dbg( ' weight smaller than 25 and link density above 0.2' ); $toRemove = true; } elseif ( $a > 2 && ($weight >= 25 && $linkDensity > 0.5) ) { $this->dbg( ' more than 2 links and weight above 25 but link density greater than 0.5' ); $toRemove = true; } elseif ( $embedCount > 3 ) { $this->dbg( ' more than 3 embeds' ); $toRemove = true; } } else { $this->dbg( 'Standard clean...' ); if ( $img > $p ) { $this->dbg( ' more image elements than paragraph elements' ); $toRemove = true; } elseif ( $li > $p && $tag != 'ul' && $tag != 'ol' ) { $this->dbg( ' too many <li> elements, and parent is not <ul> or <ol>' ); $toRemove = true; } elseif ( $input > floor( $p / 3 ) ) { $this->dbg( ' too many <input> elements' ); $toRemove = true; } elseif ( $contentLength < 25 && ($img === 0 || $img > 2) ) { $this->dbg( ' content length less than 25 chars and 0 images, or more than 2 images' ); $toRemove = true; } elseif ( $weight < 25 && $linkDensity > 0.2 ) { $this->dbg( ' weight smaller than 25 and link density above 0.2' ); $toRemove = true; } elseif ( $weight >= 25 && $linkDensity > 0.5 ) { $this->dbg( ' weight above 25 but link density greater than 0.5' ); $toRemove = true; } elseif ( ($embedCount == 1 && $contentLength < 75) || $embedCount > 1 ) { $this->dbg( ' 1 embed and content length smaller than 75 chars, or more than one embed' ); $toRemove = true; } } // var_dump($tagsList->item($i-2)); die(); if ( ( false !== stripos( $tagsList->item( $i )->textContent, 'et al' ) ) || ( 1 === preg_match( '/\([1-9]\d{3,}\)/', $tagsList->item( $i )->textContent ) ) ) { $this->dbg( ' content of element indicates reference.' ); $toRemove = false; } elseif ( ( (false != $tagsList) && method_exists( $tagsList, 'item' ) && ( $tagsList->item( $i )) && method_exists( $tagsList->item( $i ), 'parentNode' ) && ( $tagsList->item( $i )->parentNode) ) && ( ( ( 'li' === $tagsList->item( $i )->parentNode->tagName || ( (isset( $tagsList->item( $i )->parentNode->parentNode )) && ( 'li' === $tagsList->item( $i )->parentNode->parentNode->tagName ) ) ) && ( false !== stripos( $tagsList->item( $i -1 )->textContent, 'et al' ) ) ) || (( $tagsList->item( $i -1 )) && ( 1 === preg_match( '/\([1-9]\d{3,}\)/', $tagsList->item( $i -1 )->textContent ) ) ) ) ) { $this->dbg( ' content of element indicates reference.' ); $toRemove = false; } if ( $toRemove ) { // $this->dbg('Removing: '.$tagsList->item($i)->innerHTML); $tagsList->item( $i )->parentNode->removeChild( $tagsList->item( $i ) ); } } } } /** * Clean out spurious headers from an Element. Checks things like classnames and link density. * * @param DOMElement $e * @return void */ public function cleanHeaders( $e ) { for ( $headerIndex = 1; $headerIndex < 3; $headerIndex++ ) { $headers = $e->getElementsByTagName( 'h' . $headerIndex ); for ( $i = $headers->length -1; $i >= 0; $i-- ) { if ( $this->getClassWeight( $headers->item( $i ) ) < 0 || $this->getLinkDensity( $headers->item( $i ) ) > 0.33 ) { $headers->item( $i )->parentNode->removeChild( $headers->item( $i ) ); } } } } public function flagIsActive( $flag ) { return ($this->flags & $flag) > 0; } public function addFlag( $flag ) { $this->flags = $this->flags | $flag; } public function removeFlag( $flag ) { $this->flags = $this->flags & ~$flag; } }
| ver. 1.4 |
Github
|
.
| PHP 7.4.33 | Генерация страницы: 0.01 |
proxy
|
phpinfo
|
Настройка