mirror of
https://github.com/j0k3r/php-readability.git
synced 2026-09-27 14:36:23 +00:00
Compare commits
20
Commits
a8b08d8cb2
..
2.0.9
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a93488d727 | ||
|
|
ac24ef54a4 | ||
|
|
76547fef78 | ||
|
|
228bc7ee1d | ||
|
|
35b87585e7 | ||
|
|
4230b2d7ca | ||
|
|
85be584f94 | ||
|
|
03a960daf0 | ||
|
|
6f62bcf662 | ||
|
|
3e3114a492 | ||
|
|
f98247ed14 | ||
|
|
5a6e525ff5 | ||
|
|
0de328760a | ||
|
|
5aa9da6843 | ||
|
|
1bb7eec83c | ||
|
|
049bd07074 | ||
|
|
ca1b105f40 | ||
|
|
5ad159ebb1 | ||
|
|
1312ffbced | ||
|
|
266e36c187 |
+4
-4
@@ -30,12 +30,12 @@
|
|||||||
"masterminds/html5": "^2.7"
|
"masterminds/html5": "^2.7"
|
||||||
},
|
},
|
||||||
"require-dev": {
|
"require-dev": {
|
||||||
"friendsofphp/php-cs-fixer": "3.95.8",
|
"friendsofphp/php-cs-fixer": "3.95.12",
|
||||||
"monolog/monolog": "^1.24|^2.1",
|
"monolog/monolog": "^1.24|^2.1",
|
||||||
"symfony/phpunit-bridge": "^4.4|^5.3|^6.0|^7.0",
|
"symfony/phpunit-bridge": "^4.4|^5.3|^6.0|^7.0",
|
||||||
"phpstan/phpstan": "2.2.2",
|
"phpstan/phpstan": "2.2.5",
|
||||||
"phpstan/phpstan-phpunit": "2.0.16",
|
"phpstan/phpstan-phpunit": "2.0.18",
|
||||||
"rector/rector": "2.4.6"
|
"rector/rector": "2.5.5"
|
||||||
},
|
},
|
||||||
"suggest": {
|
"suggest": {
|
||||||
"ext-tidy": "Used to clean up given HTML and to avoid problems with bad HTML structure."
|
"ext-tidy": "Used to clean up given HTML and to avoid problems with bad HTML structure."
|
||||||
|
|||||||
+78
-72
@@ -213,7 +213,9 @@ class Readability implements LoggerAwareInterface
|
|||||||
*/
|
*/
|
||||||
public function init(): bool
|
public function init(): bool
|
||||||
{
|
{
|
||||||
$this->loadHtml();
|
if (!isset($this->dom)) {
|
||||||
|
$this->loadHtml();
|
||||||
|
}
|
||||||
|
|
||||||
if (!isset($this->dom->documentElement)) {
|
if (!isset($this->dom->documentElement)) {
|
||||||
return false;
|
return false;
|
||||||
@@ -751,6 +753,76 @@ class Readability implements LoggerAwareInterface
|
|||||||
$this->flags &= ~$flag;
|
$this->flags &= ~$flag;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Load HTML in a DOMDocument.
|
||||||
|
* Apply Pre filters
|
||||||
|
* Cleanup HTML using Tidy (or not).
|
||||||
|
*/
|
||||||
|
public function loadHtml(): void
|
||||||
|
{
|
||||||
|
$this->original_html = $this->html;
|
||||||
|
|
||||||
|
$this->logger->debug('Parsing URL: ' . $this->url);
|
||||||
|
|
||||||
|
if ($this->url) {
|
||||||
|
$host = parse_url($this->url, \PHP_URL_HOST);
|
||||||
|
if (null !== $host) {
|
||||||
|
$this->domainRegExp = '/' . strtr((string) preg_replace('/www\d*\./', '', $host), ['.' => '\.']) . '/';
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
mb_internal_encoding('UTF-8');
|
||||||
|
mb_http_output('UTF-8');
|
||||||
|
mb_regex_encoding('UTF-8');
|
||||||
|
|
||||||
|
// HACK: dirty cleanup to replace some stuff; shouldn't use regexps with HTML but well...
|
||||||
|
if (!$this->flagIsActive(self::FLAG_DISABLE_PREFILTER)) {
|
||||||
|
foreach ($this->pre_filters as $search => $replace) {
|
||||||
|
$this->html = preg_replace($search, $replace, $this->html);
|
||||||
|
}
|
||||||
|
unset($search, $replace);
|
||||||
|
}
|
||||||
|
|
||||||
|
if ('' === trim($this->html)) {
|
||||||
|
$this->html = '<html></html>';
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Use tidy (if it exists).
|
||||||
|
* This fixes problems with some sites which would otherwise trouble DOMDocument's HTML parsing.
|
||||||
|
* Although sometimes it makes matters worse, which is why there is an option to disable it.
|
||||||
|
*/
|
||||||
|
if ($this->useTidy) {
|
||||||
|
$this->logger->debug('Tidying document');
|
||||||
|
|
||||||
|
$tidy = tidy_repair_string($this->html, $this->tidy_config, 'UTF8');
|
||||||
|
if (false !== $tidy && $this->html !== $tidy) {
|
||||||
|
$this->tidied = true;
|
||||||
|
$this->html = $tidy;
|
||||||
|
$this->html = preg_replace('/[\r\n]+/is', "\n", $this->html);
|
||||||
|
}
|
||||||
|
unset($tidy);
|
||||||
|
}
|
||||||
|
|
||||||
|
$this->html = self::entitizeNonAscii((string) $this->html);
|
||||||
|
|
||||||
|
if ('html5lib' === $this->parser || 'html5' === $this->parser) {
|
||||||
|
$this->dom = (new HTML5())->loadHTML($this->html);
|
||||||
|
}
|
||||||
|
|
||||||
|
if ('libxml' === $this->parser) {
|
||||||
|
libxml_use_internal_errors(true);
|
||||||
|
|
||||||
|
$this->dom = new \DOMDocument();
|
||||||
|
$this->dom->preserveWhiteSpace = false;
|
||||||
|
$this->dom->loadHTML($this->html, \LIBXML_NOBLANKS | \LIBXML_COMPACT | \LIBXML_NOERROR);
|
||||||
|
|
||||||
|
libxml_use_internal_errors(false);
|
||||||
|
}
|
||||||
|
|
||||||
|
$this->dom->registerNodeClass(\DOMElement::class, JSLikeHTMLElement::class);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get the article title as an H1.
|
* Get the article title as an H1.
|
||||||
*
|
*
|
||||||
@@ -1003,7 +1075,11 @@ class Readability implements LoggerAwareInterface
|
|||||||
}
|
}
|
||||||
|
|
||||||
if ($this->hasSingleTagInsideElement($node, 'p') && $this->getLinkDensity($node) < 0.25) {
|
if ($this->hasSingleTagInsideElement($node, 'p') && $this->getLinkDensity($node) < 0.25) {
|
||||||
$newNode = $node->childNodes->item(0);
|
// In some cases when tidy is disabled the first item may not be a DOMElement so we apply a filter
|
||||||
|
$newNode = array_values(array_filter(
|
||||||
|
iterator_to_array($node->childNodes),
|
||||||
|
static fn ($childNode) => $childNode instanceof \DOMElement
|
||||||
|
))[0];
|
||||||
$node->parentNode->replaceChild($newNode, $node);
|
$node->parentNode->replaceChild($newNode, $node);
|
||||||
$nodesToScore[] = $newNode;
|
$nodesToScore[] = $newNode;
|
||||||
}
|
}
|
||||||
@@ -1371,76 +1447,6 @@ class Readability implements LoggerAwareInterface
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Load HTML in a DOMDocument.
|
|
||||||
* Apply Pre filters
|
|
||||||
* Cleanup HTML using Tidy (or not).
|
|
||||||
*/
|
|
||||||
private function loadHtml(): void
|
|
||||||
{
|
|
||||||
$this->original_html = $this->html;
|
|
||||||
|
|
||||||
$this->logger->debug('Parsing URL: ' . $this->url);
|
|
||||||
|
|
||||||
if ($this->url) {
|
|
||||||
$host = parse_url($this->url, \PHP_URL_HOST);
|
|
||||||
if (null !== $host) {
|
|
||||||
$this->domainRegExp = '/' . strtr((string) preg_replace('/www\d*\./', '', $host), ['.' => '\.']) . '/';
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
mb_internal_encoding('UTF-8');
|
|
||||||
mb_http_output('UTF-8');
|
|
||||||
mb_regex_encoding('UTF-8');
|
|
||||||
|
|
||||||
// HACK: dirty cleanup to replace some stuff; shouldn't use regexps with HTML but well...
|
|
||||||
if (!$this->flagIsActive(self::FLAG_DISABLE_PREFILTER)) {
|
|
||||||
foreach ($this->pre_filters as $search => $replace) {
|
|
||||||
$this->html = preg_replace($search, $replace, $this->html);
|
|
||||||
}
|
|
||||||
unset($search, $replace);
|
|
||||||
}
|
|
||||||
|
|
||||||
if ('' === trim($this->html)) {
|
|
||||||
$this->html = '<html></html>';
|
|
||||||
}
|
|
||||||
|
|
||||||
/*
|
|
||||||
* Use tidy (if it exists).
|
|
||||||
* This fixes problems with some sites which would otherwise trouble DOMDocument's HTML parsing.
|
|
||||||
* Although sometimes it makes matters worse, which is why there is an option to disable it.
|
|
||||||
*/
|
|
||||||
if ($this->useTidy) {
|
|
||||||
$this->logger->debug('Tidying document');
|
|
||||||
|
|
||||||
$tidy = tidy_repair_string($this->html, $this->tidy_config, 'UTF8');
|
|
||||||
if (false !== $tidy && $this->html !== $tidy) {
|
|
||||||
$this->tidied = true;
|
|
||||||
$this->html = $tidy;
|
|
||||||
$this->html = preg_replace('/[\r\n]+/is', "\n", $this->html);
|
|
||||||
}
|
|
||||||
unset($tidy);
|
|
||||||
}
|
|
||||||
|
|
||||||
$this->html = self::entitizeNonAscii((string) $this->html);
|
|
||||||
|
|
||||||
if ('html5lib' === $this->parser || 'html5' === $this->parser) {
|
|
||||||
$this->dom = (new HTML5())->loadHTML($this->html);
|
|
||||||
}
|
|
||||||
|
|
||||||
if ('libxml' === $this->parser) {
|
|
||||||
libxml_use_internal_errors(true);
|
|
||||||
|
|
||||||
$this->dom = new \DOMDocument();
|
|
||||||
$this->dom->preserveWhiteSpace = false;
|
|
||||||
$this->dom->loadHTML($this->html, \LIBXML_NOBLANKS | \LIBXML_COMPACT | \LIBXML_NOERROR);
|
|
||||||
|
|
||||||
libxml_use_internal_errors(false);
|
|
||||||
}
|
|
||||||
|
|
||||||
$this->dom->registerNodeClass(\DOMElement::class, JSLikeHTMLElement::class);
|
|
||||||
}
|
|
||||||
|
|
||||||
private function getAncestors(\DOMElement $node, int $maxDepth = 0): array
|
private function getAncestors(\DOMElement $node, int $maxDepth = 0): array
|
||||||
{
|
{
|
||||||
$ancestors = [];
|
$ancestors = [];
|
||||||
|
|||||||
@@ -507,6 +507,19 @@ class ReadabilityTest extends \PHPUnit\Framework\TestCase
|
|||||||
$this->assertSame($expected, $method->invoke($readability, $node, $tag));
|
$this->assertSame($expected, $method->invoke($readability, $node, $tag));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public function testDivToPElementWithComment(): void
|
||||||
|
{
|
||||||
|
$text = str_repeat('Padded real article text to reach decent length for content scoring threshold here. ', 8);
|
||||||
|
$html = '<div><!-- comment --><code>some code snippet</code> ' . $text . '</div>';
|
||||||
|
|
||||||
|
$readability = $this->getReadability($html, 'http://0.0.0.0', 'libxml', false);
|
||||||
|
$res = $readability->init();
|
||||||
|
|
||||||
|
$this->assertTrue($res);
|
||||||
|
$this->assertStringContainsString('some code snippet', $readability->getContent()->getInnerHtml());
|
||||||
|
$this->assertStringContainsString('Padded real article text', $readability->getContent()->getInnerHtml());
|
||||||
|
}
|
||||||
|
|
||||||
public function testKeepFootnotes(): void
|
public function testKeepFootnotes(): void
|
||||||
{
|
{
|
||||||
// from https://www.schreibdichte.de/blog/feed-aggregator-und-spaeter-lesen-dienst-im-team
|
// from https://www.schreibdichte.de/blog/feed-aggregator-und-spaeter-lesen-dienst-im-team
|
||||||
|
|||||||
Reference in New Issue
Block a user