Compare commits

...
15 Commits
Author SHA1 Message Date
Jeremy Benoist 1830dc45d4 Merge pull request #7 from j0k3r/fix-nbsp
Avoid error with  
2015-09-20 21:05:55 +02:00
Jeremy Benoist 6be1f9b984 Fix link to fivefilters fork 2015-09-18 19:19:26 +02:00
Jeremy Benoist 175196d6c2 Avoid error with  
Fix #5
2015-09-18 19:10:48 +02:00
Jeremy Benoist 2b5af601d5 Do not format output to avoid breaking apps
It'll require to jump to 2.0.0 and I think it's to soon
2015-09-15 22:25:08 +02:00
Jeremy Benoist d01eb2ac1e Use class instead of id to avoid error
It generates error like `ID XXX already defined`
2015-09-14 21:49:40 +02:00
Jeremy Benoist c5a4a490e1 CS 2015-08-24 11:10:54 +02:00
Jeremy Benoist 908a49824f Add test on title 2015-08-24 11:09:47 +02:00
Jeremy Benoist c67189248e Backport changes from wallabag
https://github.com/wallabag/php-readability/commit/e9e4ff87f8fc56d406ccdd5a9a7f1d3d6af07e79
2015-08-24 11:09:38 +02:00
Jeremy Benoist 91b80b70e2 Update HTML5 tags
From https://github.com/htacg/tidy-html5/blob/master/src/tags.c#L296
2015-08-19 23:12:31 +02:00
Jeremy Benoist 8667dae74b Merge pull request #4 from j0k3r/php53
Restore compatibility with PHP 5.3
2015-06-11 09:39:26 +02:00
Jeremy Benoist eecae93161 Try hhvm instead of nightly
HHVM nightly is no longer supported on Ubuntu Precise.
See https://github.com/travis-ci/travis-ci/issues/3788 and https://github.com/facebook/hhvm/issues/5220
2015-06-10 09:33:25 +02:00
Jeremy Benoist 814c6e4730 Restore compatibility with PHP 5.3 2015-06-10 09:33:21 +02:00
Jeremy Benoist b69619d386 Merge pull request #3 from j0k3r/phpunit-travis
Improve Travis
2015-04-29 10:27:50 +02:00
Jeremy Benoist 1963319a55 Improve Travis & add Scrutinizer
+ CS
+ Update README
2015-04-29 10:24:24 +02:00
Jeremy Benoist f5d473780d Fix javascript typo
And add coverage
2015-04-29 10:24:20 +02:00
9 changed files with 391 additions and 170 deletions
+2
View File
@@ -1 +1,3 @@
vendor/ vendor/
coverage/
composer.lock
+3
View File
@@ -0,0 +1,3 @@
tools:
external_code_coverage:
timeout: 600
+23 -2
View File
@@ -1,13 +1,34 @@
language: php language: php
php: php:
- 5.3.3
- 5.3
- 5.4 - 5.4
- 5.5 - 5.5
- 5.6 - 5.6
- nightly
- hhvm
# run build against nightly but allow them to fail
matrix:
fast_finish: true
allow_failures:
- php: nightly
- php: hhvm
# faster builds on new travis setup not using sudo
sudo: false
install:
- composer self-update
before_script: before_script:
- composer self-update
- composer install --prefer-dist --no-interaction - composer install --prefer-dist --no-interaction
script: script:
- phpunit --coverage-text - phpunit --coverage-clover=coverage.clover
after_script:
- |
wget https://scrutinizer-ci.com/ocular.phar
php ocular.phar code-coverage:upload --format=php-clover coverage.clover
+2 -1
View File
@@ -1,8 +1,9 @@
# Readability # Readability
[![Build Status](https://travis-ci.org/j0k3r/php-readability.svg?branch=master)](https://travis-ci.org/j0k3r/php-readability) [![Build Status](https://travis-ci.org/j0k3r/php-readability.svg?branch=master)](https://travis-ci.org/j0k3r/php-readability)
[![Code Coverage](https://scrutinizer-ci.com/g/j0k3r/php-readability/badges/coverage.png?b=master)](https://scrutinizer-ci.com/g/j0k3r/php-readability/?branch=master)
This is an extract of the Readability class from the [full-text-rss](https://github.com/Dither/full-text-rss) fork. It kind be defined as a better version of the original [php-readability](http://code.fivefilters.org/php-readability). This is an extract of the Readability class from the [full-text-rss](https://github.com/Dither/full-text-rss) fork. It kind be defined as a better version of the original [php-readability](https://bitbucket.org/fivefilters/php-readability/overview).
## Differences ## Differences
+1 -1
View File
@@ -24,7 +24,7 @@
"role": "Developer (original JS version)" "role": "Developer (original JS version)"
}], }],
"require": { "require": {
"php": ">=5.4", "php": ">=5.3.3",
"ext-tidy": ">=1.2" "ext-tidy": ">=1.2"
}, },
"autoload": { "autoload": {
+4 -1
View File
@@ -19,11 +19,14 @@
<filter> <filter>
<whitelist> <whitelist>
<directory>./src/TubeLink/</directory> <directory>./src/</directory>
<exclude> <exclude>
<directory>./tests</directory> <directory>./tests</directory>
</exclude> </exclude>
</whitelist> </whitelist>
</filter> </filter>
<logging>
<log type="coverage-html" target="coverage" title="Readability" charset="UTF-8" yui="true" highlight="true" lowUpperBound="35" highLowerBound="70"/>
</logging>
</phpunit> </phpunit>
+8 -5
View File
@@ -3,7 +3,7 @@
namespace Readability; namespace Readability;
/** /**
* JavaScript-like HTML DOM Element * JavaScript-like HTML DOM Element.
* *
* This class extends PHP's DOMElement to allow * This class extends PHP's DOMElement to allow
* users to get and set the innerHTML property of * users to get and set the innerHTML property of
@@ -31,12 +31,14 @@ namespace Readability;
* echo $doc->saveXML(); * echo $doc->saveXML();
* *
* @author Keyvan Minoukadeh - http://www.keyvan.net - keyvan@keyvan.net * @author Keyvan Minoukadeh - http://www.keyvan.net - keyvan@keyvan.net
*
* @see http://fivefilters.org (the project this was written for) * @see http://fivefilters.org (the project this was written for)
*/ */
class JSLikeHTMLElement extends \DOMElement class JSLikeHTMLElement extends \DOMElement
{ {
/** /**
* Used for setting innerHTML like it's done in JavaScript: * Used for setting innerHTML like it's done in JavaScript:.
*
* @code * @code
* $div->innerHTML = '<h2>Chapter 2</h2><p>The story begins...</p>'; * $div->innerHTML = '<h2>Chapter 2</h2><p>The story begins...</p>';
* @endcode * @endcode
@@ -45,7 +47,7 @@ class JSLikeHTMLElement extends \DOMElement
{ {
if ($name == 'innerHTML') { if ($name == 'innerHTML') {
// first, empty the element // first, empty the element
for ($x=$this->childNodes->length-1; $x>=0; $x--) { for ($x = $this->childNodes->length - 1; $x >= 0; --$x) {
$this->removeChild($this->childNodes->item($x)); $this->removeChild($this->childNodes->item($x));
} }
// $value holds our new inner HTML // $value holds our new inner HTML
@@ -86,7 +88,8 @@ class JSLikeHTMLElement extends \DOMElement
} }
/** /**
* Used for getting innerHTML like it's done in JavaScript: * Used for getting innerHTML like it's done in JavaScript:.
*
* @code * @code
* $string = $div->innerHTML; * $string = $div->innerHTML;
* @endcode * @endcode
@@ -105,7 +108,7 @@ class JSLikeHTMLElement extends \DOMElement
$trace = debug_backtrace(); $trace = debug_backtrace();
trigger_error('Undefined property via __get(): '.$name.' in '.$trace[0]['file'].' on line '.$trace[0]['line'], E_USER_NOTICE); trigger_error('Undefined property via __get(): '.$name.' in '.$trace[0]['file'].' on line '.$trace[0]['line'], E_USER_NOTICE);
return null; return;
} }
public function __toString() public function __toString()
+161 -107
View File
@@ -14,7 +14,7 @@ namespace Readability;
* More information: http://fivefilters.org/content-only/ * More information: http://fivefilters.org/content-only/
* License: Apache License, Version 2.0 * License: Apache License, Version 2.0
* Requires: PHP version 5.2.0+ * Requires: PHP version 5.2.0+
* Date: 2013-08-02 * Date: 2013-08-02.
* *
* Differences between the PHP port and the original * Differences between the PHP port and the original
* ------------------------------------------------------ * ------------------------------------------------------
@@ -47,11 +47,11 @@ namespace Readability;
*/ */
class Readability class Readability
{ {
public $version = '1.7.2-without-multi-page';
public $convertLinksToFootnotes = false; public $convertLinksToFootnotes = false;
public $revertForcedParagraphElements = true; public $revertForcedParagraphElements = true;
public $articleTitle; public $articleTitle;
public $articleContent; public $articleContent;
public $original_html;
public $dom; public $dom;
public $url = null; // optional - URL where HTML was retrieved public $url = null; // optional - URL where HTML was retrieved
public $lightClean = true; // preserves more content (experimental) public $lightClean = true; // preserves more content (experimental)
@@ -63,6 +63,7 @@ class Readability
protected $bodyCache = null; // Cache the body HTML in case we need to re-use it later protected $bodyCache = null; // Cache the body HTML in case we need to re-use it later
protected $flags = 7; // 1 | 2 | 4; // Start with all processing flags set. protected $flags = 7; // 1 | 2 | 4; // Start with all processing flags set.
protected $success = false; // indicates whether we were able to extract or not protected $success = false; // indicates whether we were able to extract or not
/** /**
* All of the regular expressions in use within readability. * All of the regular expressions in use within readability.
* Defined up here so we don't instantiate them repeatedly in loops. * Defined up here so we don't instantiate them repeatedly in loops.
@@ -75,7 +76,7 @@ class Readability
'divToPElements' => '/<(?:blockquote|code|div|article|footer|aside|img|p|pre|dl|ol|ul)/mi', 'divToPElements' => '/<(?:blockquote|code|div|article|footer|aside|img|p|pre|dl|ol|ul)/mi',
'killBreaks' => '/(<br\s*\/?>([ \r\n\s]|&nbsp;?)*)+/', 'killBreaks' => '/(<br\s*\/?>([ \r\n\s]|&nbsp;?)*)+/',
'media' => '!//(?:[^\.\?/]+\.)?(?:youtu(?:be)?|soundcloud|dailymotion|vimeo|pornhub|xvideos|twitvid|rutube|viddler)\.(?:com|be|org|net)/!i', 'media' => '!//(?:[^\.\?/]+\.)?(?:youtu(?:be)?|soundcloud|dailymotion|vimeo|pornhub|xvideos|twitvid|rutube|viddler)\.(?:com|be|org|net)/!i',
'skipFootnoteLink' => '/^\s*(\[?[a-z0-9]{1,2}\]?|^|edit|citation needed)\s*$/i' 'skipFootnoteLink' => '/^\s*(\[?[a-z0-9]{1,2}\]?|^|edit|citation needed)\s*$/i',
); );
public $tidy_config = array( public $tidy_config = array(
'tidy-mark' => false, 'tidy-mark' => false,
@@ -88,9 +89,9 @@ class Readability
'output-xhtml' => true, 'output-xhtml' => true,
'logical-emphasis' => true, 'logical-emphasis' => true,
'show-body-only' => false, 'show-body-only' => false,
'new-blocklevel-tags' => 'article,aside,audio,details,figcaption,figure,footer,header,hgroup,nav,section,source,summary,temp,track,video', 'new-blocklevel-tags' => 'article aside audio bdi canvas details dialog figcaption figure footer header hgroup main menu menuitem nav section source summary template track video',
'new-empty-tags' => 'command,embed,keygen,source,track,wbr', 'new-empty-tags' => 'command embed keygen source track wbr',
'new-inline-tags' => 'audio,canvas,command,datalist,embed,keygen,mark,meter,output,progress,time,video,wbr', 'new-inline-tags' => 'audio command datalist embed keygen mark menuitem meter output progress source time video wbr',
'wrap' => 0, 'wrap' => 0,
'drop-empty-paras' => true, 'drop-empty-paras' => true,
'drop-proprietary-attributes' => false, 'drop-proprietary-attributes' => false,
@@ -100,7 +101,7 @@ class Readability
// 'merge-spans' => true, // 'merge-spans' => true,
'input-encoding' => '????', 'input-encoding' => '????',
'output-encoding' => 'utf8', 'output-encoding' => 'utf8',
'hide-comments' => true 'hide-comments' => true,
); );
// raw HTML filters // raw HTML filters
protected $pre_filters = array( protected $pre_filters = array(
@@ -110,7 +111,7 @@ class Readability
'!<font[^>]*>\s*\[AD\]\s*</font>!is' => '', // HACK: firewall-filtered content '!<font[^>]*>\s*\[AD\]\s*</font>!is' => '', // HACK: firewall-filtered content
'!(<br[^>]*>[ \r\n\s]*){2,}!i' => '</p><p>', // HACK: replace linebreaks plus br's with p's '!(<br[^>]*>[ \r\n\s]*){2,}!i' => '</p><p>', // HACK: replace linebreaks plus br's with p's
//'!</?noscript>!is' => '', // replace noscripts //'!</?noscript>!is' => '', // replace noscripts
'!<(/?)font[^>]*>!is' => '<\\1span>' // replace fonts to spans '!<(/?)font[^>]*>!is' => '<\\1span>', // replace fonts to spans
); );
// output HTML filters // output HTML filters
protected $post_filters = array( protected $post_filters = array(
@@ -120,8 +121,9 @@ class Readability
"/\n+/" => "\n", //single newlines cleanup "/\n+/" => "\n", //single newlines cleanup
'!<pre[^>]*>\s*<code!is' => '<pre', // modern web... '!<pre[^>]*>\s*<code!is' => '<pre', // modern web...
'!</code>\s*</pre>!is' => '</pre>', '!</code>\s*</pre>!is' => '</pre>',
'!<[hb]r>!is' => '<\\1 />' '!<[hb]r>!is' => '<\\1 />',
); );
// flags // flags
const FLAG_STRIP_UNLIKELYS = 1; const FLAG_STRIP_UNLIKELYS = 1;
const FLAG_WEIGHT_ATTRIBUTES = 2; const FLAG_WEIGHT_ATTRIBUTES = 2;
@@ -137,12 +139,14 @@ class Readability
const MIN_ARTICLE_LENGTH = 200; const MIN_ARTICLE_LENGTH = 200;
const MIN_NODE_LENGTH = 80; const MIN_NODE_LENGTH = 80;
const MAX_LINK_DENSITY = 0.25; const MAX_LINK_DENSITY = 0.25;
/** /**
* Create instance of Readability * Create instance of Readability.
*
* @param string UTF-8 encoded string * @param string UTF-8 encoded string
* @param string (optional) URL associated with HTML (for footnotes) * @param string (optional) URL associated with HTML (for footnotes)
* @param string (optional) Which parser to use for turning raw HTML into a DOMDocument * @param string (optional) Which parser to use for turning raw HTML into a DOMDocument
* @param boolean (optional) Use tidy * @param bool (optional) Use tidy
*/ */
public function __construct($html, $url = null, $parser = 'libxml', $use_tidy = true) public function __construct($html, $url = null, $parser = 'libxml', $use_tidy = true)
{ {
@@ -153,9 +157,9 @@ class Readability
$this->domainRegExp = '/'.strtr(preg_replace('/www\d*\./', '', parse_url($url, PHP_URL_HOST)), array('.' => '\.')).'/'; $this->domainRegExp = '/'.strtr(preg_replace('/www\d*\./', '', parse_url($url, PHP_URL_HOST)), array('.' => '\.')).'/';
} }
mb_internal_encoding("UTF-8"); mb_internal_encoding('UTF-8');
mb_http_output("UTF-8"); mb_http_output('UTF-8');
mb_regex_encoding("UTF-8"); mb_regex_encoding('UTF-8');
// HACK: dirty cleanup to replace some stuff; shouldn't use regexps with HTML but well... // HACK: dirty cleanup to replace some stuff; shouldn't use regexps with HTML but well...
if (!$this->flagIsActive(self::FLAG_DISABLE_PREFILTER)) { if (!$this->flagIsActive(self::FLAG_DISABLE_PREFILTER)) {
@@ -169,7 +173,7 @@ class Readability
$html = '<html></html>'; $html = '<html></html>';
} }
/** /*
* Use tidy (if it exists). * Use tidy (if it exists).
* This fixes problems with some sites which would otherwise trouble DOMDocument's HTML parsing. * This fixes problems with some sites which would otherwise trouble DOMDocument's HTML parsing.
* Although sometimes it makes matters worse, which is why there is an option to disable it. * Although sometimes it makes matters worse, which is why there is an option to disable it.
@@ -179,7 +183,7 @@ class Readability
$this->debugText .= 'Tidying document'."\n"; $this->debugText .= 'Tidying document'."\n";
$tidy = tidy_parse_string($html, $this->tidy_config, 'UTF8'); $tidy = tidy_parse_string($html, $this->tidy_config, 'UTF8');
if (tidy_clean_repair($tidy)) { if (tidy_clean_repair($tidy)) {
$original_html = $html; $this->original_html = $html;
$this->tidied = true; $this->tidied = true;
$html = $tidy->value; $html = $tidy->value;
$html = preg_replace('/<html[^>]+>/i', '<html>', $html); $html = preg_replace('/<html[^>]+>/i', '<html>', $html);
@@ -187,36 +191,49 @@ class Readability
} }
unset($tidy); unset($tidy);
} }
$html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $html = mb_convert_encoding($html, 'HTML-ENTITIES', 'UTF-8');
if (!($parser == 'html5lib' && ($this->dom = \HTML5_Parser::parse($html)))) { if (!($parser == 'html5lib' && ($this->dom = \HTML5_Parser::parse($html)))) {
libxml_use_internal_errors(true); libxml_use_internal_errors(true);
$this->dom = new \DOMDocument(); $this->dom = new \DOMDocument();
$this->dom->preserveWhiteSpace = false; $this->dom->preserveWhiteSpace = false;
@$this->dom->loadHTML($html, LIBXML_NOBLANKS | LIBXML_COMPACT | LIBXML_NOERROR);
if (PHP_VERSION_ID >= 50400) {
$this->dom->loadHTML($html, LIBXML_NOBLANKS | LIBXML_COMPACT | LIBXML_NOERROR);
} else {
$this->dom->loadHTML($html);
}
libxml_use_internal_errors(false); libxml_use_internal_errors(false);
} }
$this->dom->registerNodeClass('DOMElement', 'Readability\JSLikeHTMLElement'); $this->dom->registerNodeClass('DOMElement', 'Readability\JSLikeHTMLElement');
} }
/** /**
* Get article title element * Get article title element.
*
* @return DOMElement * @return DOMElement
*/ */
public function getTitle() public function getTitle()
{ {
return $this->articleTitle; return $this->articleTitle;
} }
/** /**
* Get article content element * Get article content element.
*
* @return DOMElement * @return DOMElement
*/ */
public function getContent() public function getContent()
{ {
return $this->articleContent; return $this->articleContent;
} }
/** /**
* Add pre filter for raw input HTML processing * Add pre filter for raw input HTML processing.
*
* @param string RegExp for replace * @param string RegExp for replace
* @param string (optional) Replacer * @param string (optional) Replacer
*/ */
@@ -224,8 +241,10 @@ class Readability
{ {
$this->pre_filters[$filter] = $replacer; $this->pre_filters[$filter] = $replacer;
} }
/** /**
* Add post filter for raw output HTML processing * Add post filter for raw output HTML processing.
*
* @param string RegExp for replace * @param string RegExp for replace
* @param string (optional) Replacer * @param string (optional) Replacer
*/ */
@@ -233,6 +252,7 @@ class Readability
{ {
$this->post_filters[$filter] = $replacer; $this->post_filters[$filter] = $replacer;
} }
/** /**
* Runs readability. * Runs readability.
* *
@@ -243,7 +263,7 @@ class Readability
* 4. Replace the current DOM tree with the new one. * 4. Replace the current DOM tree with the new one.
* 5. Read peacefully. * 5. Read peacefully.
* *
* @return boolean true if we found content, false otherwise * @return bool true if we found content, false otherwise
*/ */
public function init() public function init()
{ {
@@ -258,7 +278,7 @@ class Readability
if ($this->bodyCache == null) { if ($this->bodyCache == null) {
$this->bodyCache = ''; $this->bodyCache = '';
foreach ($bodyElems as $bodyNode) { foreach ($bodyElems as $bodyNode) {
$this->bodyCache += $bodyNode->innerHTML; $this->bodyCache .= trim($bodyNode->innerHTML);
} }
} }
if ($bodyElems->length > 0 && $this->body == null) { if ($bodyElems->length > 0 && $this->body == null) {
@@ -273,11 +293,11 @@ class Readability
if (!$articleContent) { if (!$articleContent) {
$this->success = false; $this->success = false;
$articleContent = $this->dom->createElement('div'); $articleContent = $this->dom->createElement('div');
$articleContent->setAttribute('id', 'readability-content'); $articleContent->setAttribute('class', 'readability-content');
$articleContent->innerHTML = '<p>Sorry, Readability was unable to parse this page for content.</p>'; $articleContent->innerHTML = '<p>Sorry, Readability was unable to parse this page for content.</p>';
} }
$overlay->setAttribute('id', 'readOverlay'); $overlay->setAttribute('class', 'readOverlay');
$innerDiv->setAttribute('id', 'readInner'); $innerDiv->setAttribute('class', 'readInner');
// Glue the structure of our document together. // Glue the structure of our document together.
$innerDiv->appendChild($articleTitle); $innerDiv->appendChild($articleTitle);
$innerDiv->appendChild($articleContent); $innerDiv->appendChild($articleContent);
@@ -294,8 +314,9 @@ class Readability
return $this->success; return $this->success;
} }
/** /**
* Debug * Debug.
*/ */
protected function dbg($msg) //, $error=false) protected function dbg($msg) //, $error=false)
{ {
@@ -305,20 +326,20 @@ class Readability
} }
/** /**
* Dump debug info * Dump debug info.
*/ */
protected function dump_dbg() protected function dump_dbg()
{ {
if ($this->debug) { if ($this->debug) {
openlog("Readability PHP ", LOG_PID | LOG_PERROR, 0); openlog('Readability PHP ', LOG_PID | LOG_PERROR, 0);
syslog(6, $this->debugText); // 1 - error 6 - info syslog(6, $this->debugText); // 1 - error 6 - info
} }
} }
/** /**
* Run any post-process modifications to article content as necessary. * Run any post-process modifications to article content as necessary.
* *
* @param DOMElement * @param DOMElement
* @return void
*/ */
public function postProcessContent($articleContent) public function postProcessContent($articleContent)
{ {
@@ -326,6 +347,7 @@ class Readability
$this->addFootnotes($articleContent); $this->addFootnotes($articleContent);
} }
} }
/** /**
* Get the article title as an H1. * Get the article title as an H1.
* *
@@ -337,7 +359,9 @@ class Readability
$origTitle = ''; $origTitle = '';
try { try {
$curTitle = $origTitle = $this->getInnerText($this->dom->getElementsByTagName('title')->item(0)); $curTitle = $origTitle = $this->getInnerText($this->dom->getElementsByTagName('title')->item(0));
} catch (Exception $e) {} } catch (Exception $e) {
}
if (preg_match('/ [\|\-] /', $curTitle)) { if (preg_match('/ [\|\-] /', $curTitle)) {
$curTitle = preg_replace('/(.*)[\|\-] .*/i', '$1', $origTitle); $curTitle = preg_replace('/(.*)[\|\-] .*/i', '$1', $origTitle);
if (count(explode(' ', $curTitle)) < 3) { if (count(explode(' ', $curTitle)) < 3) {
@@ -354,24 +378,25 @@ class Readability
$curTitle = $this->getInnerText($hOnes->item(0)); $curTitle = $this->getInnerText($hOnes->item(0));
} }
} }
$curTitle = trim($curTitle); $curTitle = trim($curTitle);
if (count(explode(' ', $curTitle)) <= 4) { if (count(explode(' ', $curTitle)) <= 4) {
$curTitle = $origTitle; $curTitle = $origTitle;
} }
$articleTitle = $this->dom->createElement('h1'); $articleTitle = $this->dom->createElement('h1');
$articleTitle->innerHTML = $curTitle; $articleTitle->innerHTML = $curTitle;
return $articleTitle; return $articleTitle;
} }
/** /**
* Prepare the HTML document for readability to scrape it. * Prepare the HTML document for readability to scrape it.
* This includes things like stripping javascript, CSS, and handling terrible markup. * This includes things like stripping javascript, CSS, and handling terrible markup.
*
* @return void
*/ */
protected function prepDocument() protected function prepDocument()
{ {
/** /*
* In some cases a body element can't be found (if the HTML is totally hosed for example) * In some cases a body element can't be found (if the HTML is totally hosed for example)
* so we create a new body node and append it to the document. * so we create a new body node and append it to the document.
*/ */
@@ -379,34 +404,34 @@ class Readability
$this->body = $this->dom->createElement('body'); $this->body = $this->dom->createElement('body');
$this->dom->documentElement->appendChild($this->body); $this->dom->documentElement->appendChild($this->body);
} }
$this->body->setAttribute('id', 'readabilityBody'); $this->body->setAttribute('class', 'readabilityBody');
// Remove all style tags in head. // Remove all style tags in head.
$styleTags = $this->dom->getElementsByTagName('style'); $styleTags = $this->dom->getElementsByTagName('style');
for ($i = $styleTags->length-1; $i >= 0; $i--) { for ($i = $styleTags->length - 1; $i >= 0; --$i) {
$styleTags->item($i)->parentNode->removeChild($styleTags->item($i)); $styleTags->item($i)->parentNode->removeChild($styleTags->item($i));
} }
$linkTags = $this->dom->getElementsByTagName('link'); $linkTags = $this->dom->getElementsByTagName('link');
for ($i = $linkTags->length-1; $i >= 0; $i--) { for ($i = $linkTags->length - 1; $i >= 0; --$i) {
$linkTags->item($i)->parentNode->removeChild($linkTags->item($i)); $linkTags->item($i)->parentNode->removeChild($linkTags->item($i));
} }
} }
/** /**
* For easier reading, convert this document to have footnotes at the bottom rather than inline links. * For easier reading, convert this document to have footnotes at the bottom rather than inline links.
* @see http://www.roughtype.com/archives/2010/05/experiments_in.php
* *
* @return void * @see http://www.roughtype.com/archives/2010/05/experiments_in.php
*/ */
public function addFootnotes($articleContent) public function addFootnotes($articleContent)
{ {
$footnotesWrapper = $this->dom->createElement('footer'); $footnotesWrapper = $this->dom->createElement('footer');
$footnotesWrapper->setAttribute('id', 'readability-footnotes'); $footnotesWrapper->setAttribute('class', 'readability-footnotes');
$footnotesWrapper->innerHTML = '<h3>References</h3>'; $footnotesWrapper->innerHTML = '<h3>References</h3>';
$articleFootnotes = $this->dom->createElement('ol'); $articleFootnotes = $this->dom->createElement('ol');
$articleFootnotes->setAttribute('id', 'readability-footnotes-list'); $articleFootnotes->setAttribute('class', 'readability-footnotes-list');
$footnotesWrapper->appendChild($articleFootnotes); $footnotesWrapper->appendChild($articleFootnotes);
$articleLinks = $articleContent->getElementsByTagName('a'); $articleLinks = $articleContent->getElementsByTagName('a');
$linkCount = 0; $linkCount = 0;
for ($i = 0; $i < $articleLinks->length; $i++) { for ($i = 0; $i < $articleLinks->length; ++$i) {
$articleLink = $articleLinks->item($i); $articleLink = $articleLinks->item($i);
$footnoteLink = $articleLink->cloneNode(true); $footnoteLink = $articleLink->cloneNode(true);
$refLink = $this->dom->createElement('a'); $refLink = $this->dom->createElement('a');
@@ -419,7 +444,7 @@ class Readability
if ((strpos($articleLink->getAttribute('class'), 'readability-DoNotFootnote') !== false) || preg_match($this->regexps['skipFootnoteLink'], $linkText)) { if ((strpos($articleLink->getAttribute('class'), 'readability-DoNotFootnote') !== false) || preg_match($this->regexps['skipFootnoteLink'], $linkText)) {
continue; continue;
} }
$linkCount++; ++$linkCount;
// Add a superscript reference after the article link. // Add a superscript reference after the article link.
$refLink->setAttribute('href', '#readabilityFootnoteLink-'.$linkCount); $refLink->setAttribute('href', '#readabilityFootnoteLink-'.$linkCount);
$refLink->innerHTML = '<small><sup>['.$linkCount.']</sup></small>'; $refLink->innerHTML = '<small><sup>['.$linkCount.']</sup></small>';
@@ -445,12 +470,12 @@ class Readability
$articleContent->appendChild($footnotesWrapper); $articleContent->appendChild($footnotesWrapper);
} }
} }
/** /**
* Prepare the article node for display. Clean out any inline styles, * Prepare the article node for display. Clean out any inline styles,
* iframes, forms, strip extraneous <p> tags, etc. * iframes, forms, strip extraneous <p> tags, etc.
* *
* @param DOMElement * @param DOMElement
* @return void
*/ */
public function prepArticle($articleContent) public function prepArticle($articleContent)
{ {
@@ -463,25 +488,25 @@ class Readability
$this->killBreaks($articleContent); $this->killBreaks($articleContent);
$xpath = new \DOMXPath($articleContent->ownerDocument); $xpath = new \DOMXPath($articleContent->ownerDocument);
if ($this->revertForcedParagraphElements) { if ($this->revertForcedParagraphElements) {
/** /*
* Reverts P elements with class 'readability-styled' to text nodes: * Reverts P elements with class 'readability-styled' to text nodes:
* which is what they were before. * which is what they were before.
*/ */
$elems = $xpath->query('.//p[@data-readability-styled]', $articleContent); $elems = $xpath->query('.//p[@data-readability-styled]', $articleContent);
for ($i = $elems->length-1; $i >= 0; $i--) { for ($i = $elems->length - 1; $i >= 0; --$i) {
$e = $elems->item($i); $e = $elems->item($i);
$e->parentNode->replaceChild($articleContent->ownerDocument->createTextNode($e->textContent), $e); $e->parentNode->replaceChild($articleContent->ownerDocument->createTextNode($e->textContent), $e);
} }
} }
// Remove service data-candidate attribute. // Remove service data-candidate attribute.
$elems = $xpath->query('.//*[@data-candidate]', $articleContent); $elems = $xpath->query('.//*[@data-candidate]', $articleContent);
for ($i = $elems->length-1; $i >= 0; $i--) { for ($i = $elems->length - 1; $i >= 0; --$i) {
$elems->item($i)->removeAttribute('data-candidate'); $elems->item($i)->removeAttribute('data-candidate');
} }
// Remove unrelated links and other unneded stuff. // Remove unrelated links and other unneded stuff.
// (not(*) and not(text()[normalize-space()])) or // What's wrong here? // (not(*) and not(text()[normalize-space()])) or // What's wrong here?
$elems = $xpath->query('.//a[@rel="nofollow"]', $articleContent); $elems = $xpath->query('.//a[@rel="nofollow"]', $articleContent);
for ($i = $elems->length-1; $i >= 0; $i--) { for ($i = $elems->length - 1; $i >= 0; --$i) {
$elems->item($i)->parentNode->removeChild($elems->item($i)); $elems->item($i)->parentNode->removeChild($elems->item($i));
} }
// Clean out junk from the article content. // Clean out junk from the article content.
@@ -493,7 +518,7 @@ class Readability
$this->clean($articleContent, 'canvas'); $this->clean($articleContent, 'canvas');
$this->clean($articleContent, 'h1'); $this->clean($articleContent, 'h1');
/** /*
* If there is only one h2, they are probably using it as a main header, so remove it since we * If there is only one h2, they are probably using it as a main header, so remove it since we
* already have a header. * already have a header.
*/ */
@@ -510,7 +535,7 @@ class Readability
$this->cleanConditionally($articleContent, 'div'); $this->cleanConditionally($articleContent, 'div');
// Remove extra paragraphs. // Remove extra paragraphs.
$articleParagraphs = $articleContent->getElementsByTagName('p'); $articleParagraphs = $articleContent->getElementsByTagName('p');
for ($i = $articleParagraphs->length-1; $i >= 0; $i--) { for ($i = $articleParagraphs->length - 1; $i >= 0; --$i) {
$imgCount = $articleParagraphs->item($i)->getElementsByTagName('img')->length; $imgCount = $articleParagraphs->item($i)->getElementsByTagName('img')->length;
$embedCount = $articleParagraphs->item($i)->getElementsByTagName('embed')->length; $embedCount = $articleParagraphs->item($i)->getElementsByTagName('embed')->length;
$objectCount = $articleParagraphs->item($i)->getElementsByTagName('object')->length; $objectCount = $articleParagraphs->item($i)->getElementsByTagName('object')->length;
@@ -523,7 +548,7 @@ class Readability
// add extra text to iframe tag to avoid an auto-closing iframe and then break the html code // add extra text to iframe tag to avoid an auto-closing iframe and then break the html code
if ($iframeCount) { if ($iframeCount) {
$iframe = $articleParagraphs->item($i)->getElementsByTagName('iframe'); $iframe = $articleParagraphs->item($i)->getElementsByTagName('iframe');
$iframe->item(0)->nodeValue = '&nbsp;'; $iframe->item(0)->nodeValue = ' ';
$articleParagraphs->item($i)->parentNode->replaceChild($iframe->item(0), $articleParagraphs->item($i)); $articleParagraphs->item($i)->parentNode->replaceChild($iframe->item(0), $articleParagraphs->item($i));
} }
@@ -536,16 +561,16 @@ class Readability
} }
unset($search, $replace); unset($search, $replace);
} catch (Exception $e) { } catch (Exception $e) {
$this->dbg("Cleaning output HTML failed. Ignoring: " . $e->getMessage()); $this->dbg('Cleaning output HTML failed. Ignoring: '.$e->getMessage());
} }
} }
} }
/** /**
* Initialize a node with the readability object. Also checks the * Initialize a node with the readability object. Also checks the
* className/id for special names to add to its score. * className/id for special names to add to its score.
* *
* @param Element * @param Element
* @return void
*/ */
protected function initializeNode($node) protected function initializeNode($node)
{ {
@@ -608,6 +633,7 @@ class Readability
} }
$readability->value += $this->getWeight($node); $readability->value += $this->getWeight($node);
} }
/** /**
* grabArticle - Using a variety of metrics (content score, classname, element types), find the content that is * grabArticle - Using a variety of metrics (content score, classname, element types), find the content that is
* most likely to be the stuff a user wants to read. Then return it wrapped up in a div. * most likely to be the stuff a user wants to read. Then return it wrapped up in a div.
@@ -625,7 +651,7 @@ class Readability
$xpath = new \DOMXPath($page); $xpath = new \DOMXPath($page);
} }
$allElements = $page->getElementsByTagName('*'); $allElements = $page->getElementsByTagName('*');
for ($nodeIndex = 0; ($node = $allElements->item($nodeIndex)); $nodeIndex++) { for ($nodeIndex = 0; ($node = $allElements->item($nodeIndex)); ++$nodeIndex) {
$tagName = $node->tagName; $tagName = $node->tagName;
// Some well known site uses sections as paragraphs. // Some well known site uses sections as paragraphs.
if (strcasecmp($tagName, 'p') === 0 || strcasecmp($tagName, 'td') === 0 || strcasecmp($tagName, 'section') === 0) { if (strcasecmp($tagName, 'p') === 0 || strcasecmp($tagName, 'td') === 0 || strcasecmp($tagName, 'section') === 0) {
@@ -643,14 +669,14 @@ class Readability
//$newNode->setAttribute('class', $node->getAttribute('class')); //$newNode->setAttribute('class', $node->getAttribute('class'));
//$newNode->setAttribute('id', $node->getAttribute('id')); //$newNode->setAttribute('id', $node->getAttribute('id'));
$node = $node->parentNode->replaceChild($newNode, $node); $node = $node->parentNode->replaceChild($newNode, $node);
$nodeIndex--; --$nodeIndex;
$nodesToScore[] = $newNode; $nodesToScore[] = $newNode;
} catch (Exception $e) { } catch (Exception $e) {
$this->dbg('Could not alter div/article to p, reverting back to div: '.$e->getMessage()); $this->dbg('Could not alter div/article to p, reverting back to div: '.$e->getMessage());
} }
} else { } else {
// Will change these P elements back to text nodes after processing. // Will change these P elements back to text nodes after processing.
for ($i = 0, $il = $node->childNodes->length; $i < $il; $i++) { for ($i = 0, $il = $node->childNodes->length; $i < $il; ++$i) {
$childNode = $node->childNodes->item($i); $childNode = $node->childNodes->item($i);
if (is_object($childNode) && get_class($childNode) === 'DOMProcessingInstruction') { //executable tags (<?php or <?xml) warning if (is_object($childNode) && get_class($childNode) === 'DOMProcessingInstruction') { //executable tags (<?php or <?xml) warning
$childNode->parentNode->removeChild($childNode); $childNode->parentNode->removeChild($childNode);
@@ -667,14 +693,14 @@ class Readability
} }
} }
} }
/** /*
* Loop through all paragraphs, and assign a score to them based on how content-y they look. * Loop through all paragraphs, and assign a score to them based on how content-y they look.
* Then add their score to their parent node. * Then add their score to their parent node.
* *
* A score is determined by things like number of commas, class names, etc. * A score is determined by things like number of commas, class names, etc.
* Maybe eventually link density. * Maybe eventually link density.
*/ */
for ($pt=0, $scored = count($nodesToScore); $pt < $scored; $pt++) { for ($pt = 0, $scored = count($nodesToScore); $pt < $scored; ++$pt) {
$parentNode = $nodesToScore[$pt]->parentNode; $parentNode = $nodesToScore[$pt]->parentNode;
// No parent node? Move on... // No parent node? Move on...
if (!$parentNode) { if (!$parentNode) {
@@ -723,13 +749,13 @@ class Readability
$grandParentNode->getAttributeNode('readability')->value += $contentScore / self::GRANDPARENT_SCORE_DIVISOR; $grandParentNode->getAttributeNode('readability')->value += $contentScore / self::GRANDPARENT_SCORE_DIVISOR;
} }
} }
/** /*
* Node prepping: trash nodes that look cruddy (like ones with the class name "comment", etc). * Node prepping: trash nodes that look cruddy (like ones with the class name "comment", etc).
* This is faster to do before scoring but safer after. * This is faster to do before scoring but safer after.
*/ */
if ($this->flagIsActive(self::FLAG_STRIP_UNLIKELYS) && $xpath) { if ($this->flagIsActive(self::FLAG_STRIP_UNLIKELYS) && $xpath) {
$candidates = $xpath->query('.//*[(self::footer and count(//footer)<2) or (self::aside and count(//aside)<2)]', $page->documentElement); $candidates = $xpath->query('.//*[(self::footer and count(//footer)<2) or (self::aside and count(//aside)<2)]', $page->documentElement);
for ($node = null, $c = $candidates->length-1; $c >= 0; $c--) { for ($node = null, $c = $candidates->length - 1; $c >= 0; --$c) {
$node = $candidates->item($c); $node = $candidates->item($c);
// node should be readable but not inside of an article otherwise it's probably non-readable block // node should be readable but not inside of an article otherwise it's probably non-readable block
if ($node->hasAttribute('readability') && (int) $node->getAttributeNode('readability')->value < 40 && ($node->parentNode ? strcasecmp($node->parentNode->tagName, 'article') !== 0 : true)) { if ($node->hasAttribute('readability') && (int) $node->getAttributeNode('readability')->value < 40 && ($node->parentNode ? strcasecmp($node->parentNode->tagName, 'article') !== 0 : true)) {
@@ -738,11 +764,11 @@ class Readability
} }
} }
$candidates = $xpath->query('.//*[not(self::body) and (@class or @id or @style) and ((number(@readability) < 40) or not(@readability))]', $page->documentElement); $candidates = $xpath->query('.//*[not(self::body) and (@class or @id or @style) and ((number(@readability) < 40) or not(@readability))]', $page->documentElement);
for ($node = null, $c = $candidates->length-1; $c >= 0; $c--) { for ($node = null, $c = $candidates->length - 1; $c >= 0; --$c) {
$node = $candidates->item($c); $node = $candidates->item($c);
$tagName = $node->tagName; $tagName = $node->tagName;
/* Remove unlikely candidates */ /* Remove unlikely candidates */
$unlikelyMatchString = $node->getAttribute('class')." ".$node->getAttribute('id')." ".$node->getAttribute('style'); $unlikelyMatchString = $node->getAttribute('class').' '.$node->getAttribute('id').' '.$node->getAttribute('style');
//$this->dbg('Processing '.$node->getNodePath().' by "'. $unlikelyMatchString.'" with readability '.($node->hasAttribute('readability') ? (int)$node->getAttributeNode('readability')->value : 0)); //$this->dbg('Processing '.$node->getNodePath().' by "'. $unlikelyMatchString.'" with readability '.($node->hasAttribute('readability') ? (int)$node->getAttributeNode('readability')->value : 0));
if (mb_strlen($unlikelyMatchString) > 3 && // don't process "empty" strings if (mb_strlen($unlikelyMatchString) > 3 && // don't process "empty" strings
preg_match($this->regexps['unlikelyCandidates'], $unlikelyMatchString) && preg_match($this->regexps['unlikelyCandidates'], $unlikelyMatchString) &&
@@ -750,12 +776,12 @@ class Readability
) { ) {
$this->dbg('Removing unlikely candidate '.$node->getNodePath().' by "'.$unlikelyMatchString.'" with readability '.($node->hasAttribute('readability') ? (int) $node->getAttributeNode('readability')->value : 0)); $this->dbg('Removing unlikely candidate '.$node->getNodePath().' by "'.$unlikelyMatchString.'" with readability '.($node->hasAttribute('readability') ? (int) $node->getAttributeNode('readability')->value : 0));
$node->parentNode->removeChild($node); $node->parentNode->removeChild($node);
$nodeIndex--; --$nodeIndex;
} }
} }
unset($candidates); unset($candidates);
} }
/** /*
* After we've calculated scores, loop through all of the possible candidate nodes we found * After we've calculated scores, loop through all of the possible candidate nodes we found
* and find the one with the highest score. * and find the one with the highest score.
*/ */
@@ -763,7 +789,7 @@ class Readability
if ($xpath) { if ($xpath) {
// Using array of DOMElements after deletion is a path to DOOMElement. // Using array of DOMElements after deletion is a path to DOOMElement.
$candidates = $xpath->query('.//*[@data-candidate]', $page->documentElement); $candidates = $xpath->query('.//*[@data-candidate]', $page->documentElement);
for ($c = $candidates->length-1; $c >= 0; $c--) { for ($c = $candidates->length - 1; $c >= 0; --$c) {
// Scale the final candidates score based on link density. Good content should have a // Scale the final candidates score based on link density. Good content should have a
// relatively small link density (5% or less) and be mostly unaffected by this operation. // relatively small link density (5% or less) and be mostly unaffected by this operation.
// If not for this we would have used XPath to find maximum @readability. // If not for this we would have used XPath to find maximum @readability.
@@ -776,7 +802,7 @@ class Readability
} }
unset($candidates); unset($candidates);
} }
/** /*
* If we still have no top candidate, just use the body as a last resort. * If we still have no top candidate, just use the body as a last resort.
* We also have to copy the body node so it is something we can modify. * We also have to copy the body node so it is something we can modify.
*/ */
@@ -790,6 +816,7 @@ class Readability
$this->dbg('Setting body to a raw HTML of original page!'); $this->dbg('Setting body to a raw HTML of original page!');
$topCandidate->innerHTML = $page->documentElement->innerHTML; $topCandidate->innerHTML = $page->documentElement->innerHTML;
$page->documentElement->innerHTML = ''; $page->documentElement->innerHTML = '';
$this->reinitBody();
$page->documentElement->appendChild($topCandidate); $page->documentElement->appendChild($topCandidate);
} }
} else { } else {
@@ -811,19 +838,19 @@ class Readability
} }
} }
$this->dbg('Top candidate: '.$topCandidate->getNodePath()); $this->dbg('Top candidate: '.$topCandidate->getNodePath());
/** /*
* Now that we have the top candidate, look through its siblings for content that might also be related. * Now that we have the top candidate, look through its siblings for content that might also be related.
* Things like preambles, content split by ads that we removed, etc. * Things like preambles, content split by ads that we removed, etc.
*/ */
$articleContent = $this->dom->createElement('div'); $articleContent = $this->dom->createElement('div');
$articleContent->setAttribute('id', 'readability-content'); $articleContent->setAttribute('class', 'readability-content');
$siblingScoreThreshold = max(10, ((int) $topCandidate->getAttribute('readability')) * 0.2); $siblingScoreThreshold = max(10, ((int) $topCandidate->getAttribute('readability')) * 0.2);
$siblingNodes = $topCandidate->parentNode->childNodes; $siblingNodes = $topCandidate->parentNode->childNodes;
if (!isset($siblingNodes)) { if (!isset($siblingNodes)) {
$siblingNodes = new stdClass(); $siblingNodes = new stdClass();
$siblingNodes->length = 0; $siblingNodes->length = 0;
} }
for ($s = 0, $sl = $siblingNodes->length; $s < $sl; $s++) { for ($s = 0, $sl = $siblingNodes->length; $s < $sl; ++$s) {
$siblingNode = $siblingNodes->item($s); $siblingNode = $siblingNodes->item($s);
$siblingNodeName = $siblingNode->nodeName; $siblingNodeName = $siblingNode->nodeName;
$append = false; $append = false;
@@ -858,19 +885,22 @@ class Readability
$this->dbg('Altering siblingNode '.$siblingNodeName.' to div.'); $this->dbg('Altering siblingNode '.$siblingNodeName.' to div.');
$nodeToAppend = $this->dom->createElement('div'); $nodeToAppend = $this->dom->createElement('div');
try { try {
if ($siblingNode->getAttribute('id')) {
$nodeToAppend->setAttribute('id', $siblingNode->getAttribute('id')); $nodeToAppend->setAttribute('id', $siblingNode->getAttribute('id'));
}
$nodeToAppend->setAttribute('alt', $siblingNodeName); $nodeToAppend->setAttribute('alt', $siblingNodeName);
$nodeToAppend->innerHTML = $siblingNode->innerHTML; $nodeToAppend->innerHTML = $siblingNode->innerHTML;
} catch (Exception $e) { } catch (Exception $e) {
$this->dbg('Could not alter siblingNode '.$siblingNodeName.' to div, reverting to original.'); $this->dbg('Could not alter siblingNode '.$siblingNodeName.' to div, reverting to original.');
$nodeToAppend = $siblingNode; $nodeToAppend = $siblingNode;
$s--; --$s;
$sl--; --$sl;
} }
} else { } else {
$nodeToAppend = $siblingNode; $nodeToAppend = $siblingNode;
$s--; --$s;
$sl--; --$sl;
} }
// To ensure a node does not interfere with readability styles, remove its classnames & ids. // To ensure a node does not interfere with readability styles, remove its classnames & ids.
// Now done via RegExp post_filter. // Now done via RegExp post_filter.
@@ -883,30 +913,28 @@ class Readability
unset($xpath); unset($xpath);
// So we have all of the content that we need. Now we clean it up for presentation. // So we have all of the content that we need. Now we clean it up for presentation.
$this->prepArticle($articleContent); $this->prepArticle($articleContent);
/** /*
* Now that we've gone through the full algorithm, check to see if we got any meaningful content. * Now that we've gone through the full algorithm, check to see if we got any meaningful content.
* If we didn't, we may need to re-run grabArticle with different flags set. This gives us a higher * If we didn't, we may need to re-run grabArticle with different flags set. This gives us a higher
* likelihood of finding the content, and the sieve approach gives us a higher likelihood of * likelihood of finding the content, and the sieve approach gives us a higher likelihood of
* finding the -right- content. * finding the -right- content.
*/ */
if (mb_strlen($this->getInnerText($articleContent, false)) < self::MIN_ARTICLE_LENGTH) { if (mb_strlen($this->getInnerText($articleContent, false)) < self::MIN_ARTICLE_LENGTH) {
if (!$this->body->hasChildNodes()) { $this->reinitBody();
$this->body = $this->dom->createElement('body');
}
$this->body->innerHTML = $this->bodyCache;
if ($this->flagIsActive(self::FLAG_STRIP_UNLIKELYS)) { if ($this->flagIsActive(self::FLAG_STRIP_UNLIKELYS)) {
$this->removeFlag(self::FLAG_STRIP_UNLIKELYS); $this->removeFlag(self::FLAG_STRIP_UNLIKELYS);
$this->dbg("...content is shorter than ".self::MIN_ARTICLE_LENGTH." letters, trying not to strip unlikely content.\n"); $this->dbg('...content is shorter than '.self::MIN_ARTICLE_LENGTH." letters, trying not to strip unlikely content.\n");
return $this->grabArticle($this->body); return $this->grabArticle($this->body);
} elseif ($this->flagIsActive(self::FLAG_WEIGHT_ATTRIBUTES)) { } elseif ($this->flagIsActive(self::FLAG_WEIGHT_ATTRIBUTES)) {
$this->removeFlag(self::FLAG_WEIGHT_ATTRIBUTES); $this->removeFlag(self::FLAG_WEIGHT_ATTRIBUTES);
$this->dbg("...content is shorter than ".self::MIN_ARTICLE_LENGTH." letters, trying not to weight attributes.\n"); $this->dbg('...content is shorter than '.self::MIN_ARTICLE_LENGTH." letters, trying not to weight attributes.\n");
return $this->grabArticle($this->body); return $this->grabArticle($this->body);
} elseif ($this->flagIsActive(self::FLAG_CLEAN_CONDITIONALLY)) { } elseif ($this->flagIsActive(self::FLAG_CLEAN_CONDITIONALLY)) {
$this->removeFlag(self::FLAG_CLEAN_CONDITIONALLY); $this->removeFlag(self::FLAG_CLEAN_CONDITIONALLY);
$this->dbg("...content is shorter than ".self::MIN_ARTICLE_LENGTH." letters, trying not to clean at all.\n"); $this->dbg('...content is shorter than '.self::MIN_ARTICLE_LENGTH." letters, trying not to clean at all.\n");
return $this->grabArticle($this->body); return $this->grabArticle($this->body);
} else { } else {
@@ -916,13 +944,15 @@ class Readability
return $articleContent; return $articleContent;
} }
/** /**
* Get the inner text of a node. * Get the inner text of a node.
* This also strips out any excess whitespace to be found. * This also strips out any excess whitespace to be found.
* *
* @param DOMElement $e * @param DOMElement $e
* @param boolean $normalizeSpaces (default: true) * @param bool $normalizeSpaces (default: true)
* @param boolean $flattenLines (default: false) * @param bool $flattenLines (default: false)
*
* @return string * @return string
*/ */
public function getInnerText($e, $normalizeSpaces = true, $flattenLines = false) public function getInnerText($e, $normalizeSpaces = true, $flattenLines = false)
@@ -939,11 +969,11 @@ class Readability
return $textContent; return $textContent;
} }
/** /**
* Remove the style attribute on every $e and under. * Remove the style attribute on every $e and under.
* *
* @param DOMElement $e * @param DOMElement $e
* @return void
*/ */
public function cleanStyles($e) public function cleanStyles($e)
{ {
@@ -955,27 +985,32 @@ class Readability
$elem->removeAttribute('style'); $elem->removeAttribute('style');
} }
} }
/** /**
* Get comma number for a given text. * Get comma number for a given text.
* *
* @param string $text * @param string $text
*
* @return number (integer) * @return number (integer)
*/ */
public function getCommaCount($text) public function getCommaCount($text)
{ {
return substr_count($text, ','); return substr_count($text, ',');
} }
/** /**
* Get words number for a given text if words separated by a space. * Get words number for a given text if words separated by a space.
* Input string should be normalized. * Input string should be normalized.
* *
* @param string $text * @param string $text
*
* @return number (integer) * @return number (integer)
*/ */
public function getWordCount($text) public function getWordCount($text)
{ {
return substr_count($text, ' '); return substr_count($text, ' ');
} }
/** /**
* Get the density of links as a percentage of the content * Get the density of links as a percentage of the content
* This is the amount of text that is inside a link divided by the total text in the node. * This is the amount of text that is inside a link divided by the total text in the node.
@@ -983,6 +1018,7 @@ class Readability
* *
* @param DOMElement $e * @param DOMElement $e
* @param string $excludeExternal * @param string $excludeExternal
*
* @return number (float) * @return number (float)
*/ */
public function getLinkDensity($e, $excludeExternal = false) public function getLinkDensity($e, $excludeExternal = false)
@@ -990,7 +1026,7 @@ class Readability
$links = $e->getElementsByTagName('a'); $links = $e->getElementsByTagName('a');
$textLength = mb_strlen($this->getInnerText($e, true, true)); $textLength = mb_strlen($this->getInnerText($e, true, true));
$linkLength = 0; $linkLength = 0;
for ($dRe = $this->domainRegExp, $i=0, $il=$links->length; $i < $il; $i++) { for ($dRe = $this->domainRegExp, $i = 0, $il = $links->length; $i < $il; ++$i) {
if ($excludeExternal && $dRe && !preg_match($dRe, $links->item($i)->getAttribute('href'))) { if ($excludeExternal && $dRe && !preg_match($dRe, $links->item($i)->getAttribute('href'))) {
continue; continue;
} }
@@ -1002,12 +1038,14 @@ class Readability
return 0; return 0;
} }
} }
/** /**
* Get an element weight by attribute. * Get an element weight by attribute.
* Uses regular expressions to tell if this element looks good or bad. * Uses regular expressions to tell if this element looks good or bad.
* *
* @param DOMElement $element * @param DOMElement $element
* @param string $attribute * @param string $attribute
*
* @return number (Integer) * @return number (Integer)
*/ */
protected function weightAttribute($element, $attribute) protected function weightAttribute($element, $attribute)
@@ -1035,10 +1073,12 @@ class Readability
return $weight; return $weight;
} }
/** /**
* Get an element relative weight. * Get an element relative weight.
* *
* @param DOMElement $e * @param DOMElement $e
*
* @return number (Integer) * @return number (Integer)
*/ */
public function getWeight($e) public function getWeight($e)
@@ -1054,11 +1094,11 @@ class Readability
return $weight; return $weight;
} }
/** /**
* Remove extraneous break tags from a node. * Remove extraneous break tags from a node.
* *
* @param DOMElement $node * @param DOMElement $node
* @return void
*/ */
public function killBreaks($node) public function killBreaks($node)
{ {
@@ -1066,21 +1106,21 @@ class Readability
$html = preg_replace($this->regexps['killBreaks'], '<br />', $html); $html = preg_replace($this->regexps['killBreaks'], '<br />', $html);
$node->innerHTML = $html; $node->innerHTML = $html;
} }
/** /**
* Clean a node of all elements of type "tag". * Clean a node of all elements of type "tag".
* (Unless it's a youtube/vimeo video. People love movies.) * (Unless it's a youtube/vimeo video. People love movies.).
* *
* Updated 2012-09-18 to preserve youtube/vimeo iframes * Updated 2012-09-18 to preserve youtube/vimeo iframes
* *
* @param DOMElement $e * @param DOMElement $e
* @param string $tag * @param string $tag
* @return void
*/ */
public function clean($e, $tag) public function clean($e, $tag)
{ {
$targetList = $e->getElementsByTagName($tag); $targetList = $e->getElementsByTagName($tag);
$isEmbed = ($tag === 'audio' || $tag === 'video' || $tag === 'iframe' || $tag === 'object' || $tag === 'embed'); $isEmbed = ($tag === 'audio' || $tag === 'video' || $tag === 'iframe' || $tag === 'object' || $tag === 'embed');
for ($cur_item = null, $y = $targetList->length-1; $y >= 0; $y--) { for ($cur_item = null, $y = $targetList->length - 1; $y >= 0; --$y) {
/* Allow youtube and vimeo videos through as people usually want to see those. */ /* Allow youtube and vimeo videos through as people usually want to see those. */
$cur_item = $targetList->item($y); $cur_item = $targetList->item($y);
if ($isEmbed) { if ($isEmbed) {
@@ -1097,6 +1137,7 @@ class Readability
$cur_item->parentNode->removeChild($cur_item); $cur_item->parentNode->removeChild($cur_item);
} }
} }
/** /**
* Clean an element of all tags of type "tag" if they look fishy. * Clean an element of all tags of type "tag" if they look fishy.
* "Fishy" is an algorithm based on content length, classnames, * "Fishy" is an algorithm based on content length, classnames,
@@ -1104,7 +1145,6 @@ class Readability
* *
* @param DOMElement $e * @param DOMElement $e
* @param string $tag * @param string $tag
* @return void
*/ */
public function cleanConditionally($e, $tag) public function cleanConditionally($e, $tag)
{ {
@@ -1113,13 +1153,13 @@ class Readability
} }
$tagsList = $e->getElementsByTagName($tag); $tagsList = $e->getElementsByTagName($tag);
$curTagsLength = $tagsList->length; $curTagsLength = $tagsList->length;
/** /*
* Gather counts for other typical elements embedded within. * Gather counts for other typical elements embedded within.
* Traverse backwards so we can remove nodes at the same time without effecting the traversal. * Traverse backwards so we can remove nodes at the same time without effecting the traversal.
* *
* TODO: Consider taking into account original contentScore here. * TODO: Consider taking into account original contentScore here.
*/ */
for ($node = null, $i = $curTagsLength - 1; $i >= 0; $i--) { for ($node = null, $i = $curTagsLength - 1; $i >= 0; --$i) {
$node = $tagsList->item($i); $node = $tagsList->item($i);
//$class = $node->getAttribute('class').' '.$node->getAttribute('id'); //debug //$class = $node->getAttribute('class').' '.$node->getAttribute('id'); //debug
$weight = $this->getWeight($node); $weight = $this->getWeight($node);
@@ -1129,7 +1169,7 @@ class Readability
$this->dbg('Removing...'); $this->dbg('Removing...');
$node->parentNode->removeChild($node); $node->parentNode->removeChild($node);
} elseif ($this->getCommaCount($this->getInnerText($node)) < self::MIN_COMMAS_IN_PARAGRAPH) { } elseif ($this->getCommaCount($this->getInnerText($node)) < self::MIN_COMMAS_IN_PARAGRAPH) {
/** /*
* If there are not very many commas, and the number of * If there are not very many commas, and the number of
* non-paragraph elements is more than paragraphs or other ominous signs, remove the element. * non-paragraph elements is more than paragraphs or other ominous signs, remove the element.
*/ */
@@ -1140,15 +1180,15 @@ class Readability
$a = $node->getElementsByTagName('a')->length; $a = $node->getElementsByTagName('a')->length;
$embedCount = 0; $embedCount = 0;
$embeds = $node->getElementsByTagName('embed'); $embeds = $node->getElementsByTagName('embed');
for ($ei=0, $il=$embeds->length; $ei < $il; $ei++) { for ($ei = 0, $il = $embeds->length; $ei < $il; ++$ei) {
if (preg_match($this->regexps['media'], $embeds->item($ei)->getAttribute('src'))) { if (preg_match($this->regexps['media'], $embeds->item($ei)->getAttribute('src'))) {
$embedCount++; ++$embedCount;
} }
} }
$embeds = $node->getElementsByTagName('iframe'); $embeds = $node->getElementsByTagName('iframe');
for ($ei=0, $il=$embeds->length; $ei < $il; $ei++) { for ($ei = 0, $il = $embeds->length; $ei < $il; ++$ei) {
if (preg_match($this->regexps['media'], $embeds->item($ei)->getAttribute('src'))) { if (preg_match($this->regexps['media'], $embeds->item($ei)->getAttribute('src'))) {
$embedCount++; ++$embedCount;
} }
} }
$linkDensity = $this->getLinkDensity($node, true); $linkDensity = $this->getLinkDensity($node, true);
@@ -1165,10 +1205,10 @@ class Readability
$this->dbg(' content length less than 6 chars, 0 embeds and either 0 images or more than 2 images'); $this->dbg(' content length less than 6 chars, 0 embeds and either 0 images or more than 2 images');
$toRemove = true; $toRemove = true;
} elseif ($weight < 25 && $linkDensity > 0.25) { } elseif ($weight < 25 && $linkDensity > 0.25) {
$this->dbg(' weight is '.$weight.' < 25 and link density is '.sprintf("%.2f", $linkDensity).' > 0.25'); $this->dbg(' weight is '.$weight.' < 25 and link density is '.sprintf('%.2f', $linkDensity).' > 0.25');
$toRemove = true; $toRemove = true;
} elseif ($a > 2 && ($weight >= 25 && $linkDensity > 0.5)) { } elseif ($a > 2 && ($weight >= 25 && $linkDensity > 0.5)) {
$this->dbg(' more than 2 links and weight is '.$weight.' > 25 but link density is '.sprintf("%.2f", $linkDensity).' > 0.5'); $this->dbg(' more than 2 links and weight is '.$weight.' > 25 but link density is '.sprintf('%.2f', $linkDensity).' > 0.5');
$toRemove = true; $toRemove = true;
} elseif ($embedCount > 3) { } elseif ($embedCount > 3) {
$this->dbg(' more than 3 embeds'); $this->dbg(' more than 3 embeds');
@@ -1184,14 +1224,14 @@ class Readability
} elseif ($input > floor($p / 3)) { } elseif ($input > floor($p / 3)) {
$this->dbg(' too many <input> elements'); $this->dbg(' too many <input> elements');
$toRemove = true; $toRemove = true;
} elseif ($contentLength < 25 && ($img === 0 || $img > 2) ) { } elseif ($contentLength < 10 && ($img === 0 || $img > 2)) {
$this->dbg(' content length less than 25 chars and 0 images, or more than 2 images'); $this->dbg(' content length less than 10 chars and 0 images, or more than 2 images');
$toRemove = true; $toRemove = true;
} elseif ($weight < 25 && $linkDensity > 0.2) { } elseif ($weight < 25 && $linkDensity > 0.2) {
$this->dbg(' weight is '.$weight.' lower than 0 and link density is '.sprintf("%.2f", $linkDensity).' > 0.2'); $this->dbg(' weight is '.$weight.' lower than 0 and link density is '.sprintf('%.2f', $linkDensity).' > 0.2');
$toRemove = true; $toRemove = true;
} elseif ($weight >= 25 && $linkDensity > 0.5) { } elseif ($weight >= 25 && $linkDensity > 0.5) {
$this->dbg(' weight above 25 but link density is '.sprintf("%.2f", $linkDensity).' > 0.5'); $this->dbg(' weight above 25 but link density is '.sprintf('%.2f', $linkDensity).' > 0.5');
$toRemove = true; $toRemove = true;
} elseif (($embedCount == 1 && $contentLength < 75) || $embedCount > 1) { } elseif (($embedCount == 1 && $contentLength < 75) || $embedCount > 1) {
$this->dbg(' 1 embed and content length smaller than 75 chars, or more than one embed'); $this->dbg(' 1 embed and content length smaller than 75 chars, or more than one embed');
@@ -1206,33 +1246,47 @@ class Readability
} }
} }
} }
/** /**
* Clean out spurious headers from an Element. Checks things like classnames and link density. * Clean out spurious headers from an Element. Checks things like classnames and link density.
* *
* @param DOMElement $e * @param DOMElement $e
* @return void
*/ */
public function cleanHeaders($e) public function cleanHeaders($e)
{ {
for ($headerIndex = 1; $headerIndex < 3; $headerIndex++) { for ($headerIndex = 1; $headerIndex < 3; ++$headerIndex) {
$headers = $e->getElementsByTagName('h'.$headerIndex); $headers = $e->getElementsByTagName('h'.$headerIndex);
for ($i=$headers->length-1; $i >=0; $i--) { for ($i = $headers->length - 1; $i >= 0; --$i) {
if ($this->getWeight($headers->item($i)) < 0 || $this->getLinkDensity($headers->item($i)) > 0.33) { if ($this->getWeight($headers->item($i)) < 0 || $this->getLinkDensity($headers->item($i)) > 0.33) {
$headers->item($i)->parentNode->removeChild($headers->item($i)); $headers->item($i)->parentNode->removeChild($headers->item($i));
} }
} }
} }
} }
public function flagIsActive($flag) public function flagIsActive($flag)
{ {
return ($this->flags & $flag) > 0; return ($this->flags & $flag) > 0;
} }
public function addFlag($flag) public function addFlag($flag)
{ {
$this->flags = $this->flags | $flag; $this->flags = $this->flags | $flag;
} }
public function removeFlag($flag) public function removeFlag($flag)
{ {
$this->flags = $this->flags & ~$flag; $this->flags = $this->flags & ~$flag;
} }
/**
* Will recreate previously deleted body property.
*/
protected function reinitBody()
{
if (!isset($this->body->childNodes)) {
$this->body = $this->dom->createElement('body');
$this->body->innerHTML = $this->bodyCache;
}
}
} }
+135 -1
View File
@@ -3,7 +3,6 @@
namespace Tests\Readability; namespace Tests\Readability;
use Readability\Readability; use Readability\Readability;
use Readability\JSLikeHTMLElement;
class ReadabilityTested extends Readability class ReadabilityTested extends Readability
{ {
@@ -49,6 +48,8 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertFalse($res); $this->assertFalse($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('Sorry, Readability was unable to parse this page for content.', $readability->getContent()->innerHTML); $this->assertContains('Sorry, Readability was unable to parse this page for content.', $readability->getContent()->innerHTML);
} }
@@ -59,7 +60,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('<div readability=', $readability->getContent()->innerHTML); $this->assertContains('<div readability=', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is the awesome content :)', $readability->getContent()->innerHTML); $this->assertContains('This is the awesome content :)', $readability->getContent()->innerHTML);
} }
@@ -70,7 +73,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('<div readability=', $readability->getContent()->innerHTML); $this->assertContains('<div readability=', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is the awesome content :)', $readability->getContent()->innerHTML); $this->assertContains('This is the awesome content :)', $readability->getContent()->innerHTML);
} }
@@ -82,7 +87,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('<div readability=', $readability->getContent()->innerHTML); $this->assertContains('<div readability=', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is the awesome content :)', $readability->getContent()->innerHTML); $this->assertContains('This is the awesome content :)', $readability->getContent()->innerHTML);
} }
@@ -95,7 +102,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('<div readability=', $readability->getContent()->innerHTML); $this->assertContains('<div readability=', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertContains('readabilityFootnoteLink', $readability->getContent()->innerHTML); $this->assertContains('readabilityFootnoteLink', $readability->getContent()->innerHTML);
$this->assertContains('readabilityLink-3', $readability->getContent()->innerHTML); $this->assertContains('readabilityLink-3', $readability->getContent()->innerHTML);
@@ -110,7 +119,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('<div readability=', $readability->getContent()->innerHTML); $this->assertContains('<div readability=', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('will be removed', $readability->getContent()->innerHTML); $this->assertNotContains('will be removed', $readability->getContent()->innerHTML);
$this->assertNotContains('<h2>', $readability->getContent()->innerHTML); $this->assertNotContains('<h2>', $readability->getContent()->innerHTML);
@@ -124,7 +135,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('<div readability=', $readability->getContent()->innerHTML); $this->assertContains('<div readability=', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertContains('nofollow', $readability->getContent()->innerHTML); $this->assertContains('nofollow', $readability->getContent()->innerHTML);
} }
@@ -137,7 +150,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML); $this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertContains('nofollow', $readability->getContent()->innerHTML); $this->assertContains('nofollow', $readability->getContent()->innerHTML);
} }
@@ -150,7 +165,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML); $this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('<aside>', $readability->getContent()->innerHTML); $this->assertNotContains('<aside>', $readability->getContent()->innerHTML);
$this->assertContains('<footer/>', $readability->getContent()->innerHTML); $this->assertContains('<footer/>', $readability->getContent()->innerHTML);
@@ -164,7 +181,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML); $this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('This text should be removed', $readability->getContent()->innerHTML); $this->assertNotContains('This text should be removed', $readability->getContent()->innerHTML);
} }
@@ -177,7 +196,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="tr"', $readability->getContent()->innerHTML); $this->assertContains('alt="tr"', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
} }
@@ -189,7 +210,9 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML); $this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML); $this->assertContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML);
} }
@@ -202,7 +225,69 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
$this->assertTrue($res); $this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent()); $this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML); $this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEmpty($readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML);
}
public function testTitle()
{
$readability = new ReadabilityTested('<title>this is my title</title><article class="awesomecontent">'.str_repeat('<p>This is an awesome text with some links, here there are the awesome</p>', 7).'<p></p></article>', 'http://0.0.0.0');
$readability->debug = true;
$res = $readability->init();
$this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEquals('this is my title', $readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML);
}
public function testTitleWithDash()
{
$readability = new ReadabilityTested('<title> title2 - title3 </title><article class="awesomecontent">'.str_repeat('<p>This is an awesome text with some links, here there are the awesome</p>', 7).'<p></p></article>', 'http://0.0.0.0');
$readability->debug = true;
$res = $readability->init();
$this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEquals('title2 - title3', $readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML);
}
public function testTitleWithDoubleDot()
{
$readability = new ReadabilityTested('<title> title2 : title3 </title><article class="awesomecontent">'.str_repeat('<p>This is an awesome text with some links, here there are the awesome</p>', 7).'<p></p></article>', 'http://0.0.0.0');
$readability->debug = true;
$res = $readability->init();
$this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEquals('title2 : title3', $readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML);
}
public function testTitleTooShortUseH1()
{
$readability = new ReadabilityTested('<title>too short</title><h1>this is my h1 title !</h1><article class="awesomecontent">'.str_repeat('<p>This is an awesome text with some links, here there are the awesome</p>', 7).'<p></p></article>', 'http://0.0.0.0');
$readability->debug = true;
$res = $readability->init();
$this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('alt="article"', $readability->getContent()->innerHTML);
$this->assertEquals('this is my h1 title !', $readability->getTitle()->innerHTML);
$this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML); $this->assertContains('This is an awesome text with some links, here there are', $readability->getContent()->innerHTML);
$this->assertNotContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML); $this->assertNotContains('This text is also an awesome text and you should know that', $readability->getContent()->innerHTML);
} }
@@ -217,4 +302,53 @@ class ReadabilityTest extends \PHPUnit_Framework_TestCase
// $this->assertEquals('/0\.0\.0\.0/', $readability->getDomainRegexp()); // $this->assertEquals('/0\.0\.0\.0/', $readability->getDomainRegexp());
// $this->assertInstanceOf('DomDocument', $readability->dom); // $this->assertInstanceOf('DomDocument', $readability->dom);
// } // }
// dummy function to be used to the next test
public function error2Exception($code, $string, $file, $line, $context)
{
throw new \Exception($string, $code);
}
public function testAutoClosingIframeNotThrowingException()
{
error_reporting(E_ALL | E_STRICT);
ini_set('display_errors', true);
set_error_handler(array($this, 'error2Exception'), E_ALL | E_STRICT);
$data = '<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" lang="ru-RU" prefix="og: http://ogp.me/ns#">
<head profile="http://gmpg.org/xfn/11">
<meta http-equiv="Content-Type" content="text/html; charset=UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
</head>
<body class="single single-post postid-22030 single-format-standard">
<div id="wrapper">
<div id="content">
<div class="post-22030 post type-post status-publish format-standard has-post-thumbnail hentry category-video category-reviews tag-193" id="post-22030">
<h1>3D Touch &#8212; будущее мобильных игр</h1>
<div class="postdate">Автор: <strong>Сергей Пак</strong> | Просмотров: 1363 | Опубликовано: 14 сентября 2015 </div>
<div class="entry">
<p>Компания Apple представила новую технологию 3D Touch, которая является прямым потомком более ранней версии Force Touch &#8212; последняя, напомним, используется сейчас в трекпадах Macbook Pro и Macbook 2015. Теперь управлять устройством стало в разы проще, и Force Touch открывает перед пользователями новые возможности, но при этом 3D Touch &#8212; это про другое. Дело в том, что теперь и на мобильных устройствах интерфейс будет постепенно меняться, кардинальные перемены ждут мобильный гейминг, потому что здесь разработчики действительно могут разгуляться.<span id="more-22030"></span></p>
<p><iframe src="https://www.youtube.com/embed/PUep6xNeKjA" width="560" height="315" frameborder="0" allowfullscreen="allowfullscreen"></iframe></p>
<p>Итак, просто представьте себе, что iPhone 6S &#8212; это, по большому счету, отличная игровая приставка, которую вы носите с собой, а еще она может выдавать невероятной красоты картинку. Но проблема заключается, пожалуй, в том, что управлять персонажем в играх довольно трудно &#8212; он неповоротлив, обладает заторможенной реакцией, а игровой клиент зачастую требует перегруза интерфейса для того, чтобы обеспечить максимально большое количество возможностей. Благодаря трехуровневому нажатию можно избавиться от лишних кнопок и обеспечить более качественный обзор местности, и при этом пользователь будет закрывать пальцами минимальное пространство.</p>
</div>
</div>
</div>
</div>
</body>
</html>';
$readability = new ReadabilityTested($data, 'http://iosgames.ru/?p=22030');
$readability->debug = true;
$res = $readability->init();
$this->assertTrue($res);
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getContent());
$this->assertInstanceOf('Readability\JSLikeHTMLElement', $readability->getTitle());
$this->assertContains('<iframe src="https://www.youtube.com/embed/PUep6xNeKjA" width="560" height="315" frameborder="0" allowfullscreen="allowfullscreen"> </iframe>', $readability->getContent()->innerHTML);
$this->assertContains('3D Touch', $readability->getTitle()->innerHTML);
}
} }