* @license MIT */ declare(strict_types=1); namespace FeedReaderCentral; use BitBadger\InspiredByFSharp\Result; use DOMDocument; use DOMElement; use DOMException; use DOMNode; /** * A feed, as parsed from the Atom or RSS XML */ readonly class ParsedFeed { /** * Constructor * * @param string $url The URL for the feed * @param string $title The title of the feed * @param string|null $updatedOn When the feed was last updated * @param ParsedItem[] $items The items contained in the feed */ public function __construct(public string $url = '', public string $title = '', public ?string $updatedOn = null, public array $items = []) {} /** @var string The XML namespace for Atom feeds */ public const ATOM_NS = 'http://www.w3.org/2005/Atom'; /** @var string The XML namespace for the `` tag that allows HTML content in a feed */ public const CONTENT_NS = 'http://purl.org/rss/1.0/modules/content/'; /** @var string The XML namespace for XHTML */ public const XHTML_NS = 'http://www.w3.org/1999/xhtml'; /** @var string The user agent for Feed Reader Central's refresh requests */ private const USER_AGENT = 'FeedReaderCentral/' . FRC_VERSION . ' +https://bitbadger.solutions/open-source/feed-reader-central'; /** * When parsing XML into a DOMDocument, errors are presented as warnings; this creates an exception for them * * @param int $errno The error level encountered * @param string $errstr The text of the error encountered * @return bool False, to delegate to the next error handler in the chain * @throws DOMException If the error is a warning */ private static function xmlParseError(int $errno, string $errstr): bool { if ($errno === E_WARNING && substr_count($errstr, 'DOMDocument::loadXML()') > 0) { throw new DOMException($errstr, $errno); } return false; } /** * Parse a feed into an XML tree * * @param string $content The feed's RSS content * @return Result The feed if successful, an error message if not */ public static function parseFeed(string $content): Result { set_error_handler(self::xmlParseError(...)); try { $feed = new DOMDocument(); $feed->loadXML($content); return Result::OK($feed); } catch (DOMException $ex) { return Result::Error($ex->getMessage()); } finally { restore_error_handler(); } } /** * Get the value of a child element by its tag name for an RSS feed * * @param DOMNode $element The parent element * @param string $tagName The name of the tag whose value should be obtained * @return string The value of the element (or "[element] not found" if that element does not exist) */ public static function rssValue(DOMNode $element, string $tagName): string { $tags = $element->getElementsByTagName($tagName); return $tags->length === 0 ? "$tagName not found" : $tags->item(0)->textContent; } /** * Extract items from an RSS feed * * @param DOMDocument $xml The XML received from the feed * @param string $url The actual URL for the feed * @return Result The feed if successful, an error message if not */ private static function fromRSS(DOMDocument $xml, string $url): Result { $channel = $xml->getElementsByTagName('channel')->item(0); if (!($channel instanceof DOMElement)) { $type = $channel?->nodeType ?? -1; return Result::Error("Channel element not found ($type)"); } // The Atom namespace provides a lastBuildDate, which contains the last time an item in the feed was updated; if // that is not present, use the pubDate element instead if (($updatedOn = self::rssValue($channel, 'lastBuildDate')) == 'lastBuildDate not found') { if (($updatedOn = self::rssValue($channel, 'pubDate')) == 'pubDate not found') { $updatedOn = null; } } return Result::OK(new self( url: $url, title: self::rssValue($channel, 'title'), updatedOn: Data::formatDate($updatedOn), items: array_map(ParsedItem::fromRSS(...), iterator_to_array($channel->getElementsByTagName('item'))))); } /** * Get an attribute value from a DOM node * * @param DOMNode $node The node with an attribute value to obtain * @param string $name The name of the attribute whose value should be obtained * @return string The attribute value if it exists, an empty string if not */ private static function attrValue(DOMNode $node, string $name): string { return ($node->hasAttributes() ? $node->attributes->getNamedItem($name)?->value : null) ?? ''; } /** * Get the value of a child element by its tag name for an Atom feed * * (Atom feeds can have type attributes on nearly any value. For our purposes, types "text" and "html" will work as * regular string values; for "xhtml", though, we will need to get the `
` and extract its contents instead.) * * @param DOMNode $element The parent element * @param string $tagName The name of the tag whose value should be obtained * @return string The value of the element (or "[element] not found" if that element does not exist) */ public static function atomValue(DOMNode $element, string $tagName): string { $tags = $element->getElementsByTagName($tagName); if ($tags->length === 0) return "$tagName not found"; $tag = $tags->item(0); if (!($tag instanceof DOMElement)) return $tag->textContent; if (self::attrValue($tag, 'type') == 'xhtml') { $div = $tag->getElementsByTagNameNS(self::XHTML_NS, 'div'); if ($div->length === 0) return "-- invalid XHTML content --"; return $div->item(0)->textContent; } return $tag->textContent; } /** * Extract items from an Atom feed * * @param DOMDocument $xml The XML received from the feed * @param string $url The actual URL for the feed * @return Result The feed (does not have any error handling) */ private static function fromAtom(DOMDocument $xml, string $url): Result { $root = $xml->getElementsByTagNameNS(self::ATOM_NS, 'feed')->item(0); if (($updatedOn = self::atomValue($root, 'updated')) == 'pubDate not found') $updatedOn = null; return Result::OK(new self( url: $url, title: self::atomValue($root, 'title'), updatedOn: Data::formatDate($updatedOn), items: array_map(ParsedItem::fromAtom(...), iterator_to_array($root->getElementsByTagName('entry'))))); } /** * Retrieve a document (http/https) * * @param string $url The URL of the document to retrieve * @return Result ['content' => doc content, 'code' => HTTP response code, 'url' => effective URL] if * successful, an error message if not */ private static function retrieveDocument(string $url): Result { $docReq = curl_init($url); try { curl_setopt($docReq, CURLOPT_FOLLOWLOCATION, true); curl_setopt($docReq, CURLOPT_RETURNTRANSFER, true); curl_setopt($docReq, CURLOPT_CONNECTTIMEOUT, 5); curl_setopt($docReq, CURLOPT_TIMEOUT, 15); curl_setopt($docReq, CURLOPT_USERAGENT, self::USER_AGENT); $error = curl_error($docReq); if ($error !== '') return Result::Error($error); return Result::OK([ 'content' => curl_exec($docReq), 'code' => curl_getinfo($docReq, CURLINFO_RESPONSE_CODE), 'url' => curl_getinfo($docReq, CURLINFO_EFFECTIVE_URL) ]); } finally { curl_close($docReq); } } /** * Derive a feed URL from an HTML document * * @param string $content The HTML document content from which to derive a feed URL * @return Result The feed URL if successful, an error message if not */ private static function deriveFeedFromHTML(string $content): Result { $html = new DOMDocument(); $html->loadHTML(substr($content, 0, strpos($content, '') + 7)); $headTags = $html->getElementsByTagName('head'); if ($headTags->length < 1) return Result::Error('Cannot find feed at this URL'); $head = $headTags->item(0); foreach ($head->getElementsByTagName('link') as $link) { if (self::attrValue($link, 'rel') === 'alternate') { $type = self::attrValue($link, 'type'); if ($type === 'application/rss+xml' || $type === 'application/atom+xml') { return Result::OK(self::attrValue($link, 'href')); } } } return Result::Error('Cannot find feed at this URL'); } /** * Retrieve the feed * * @param string $url The URL of the feed to retrieve * @return Result The feed if successful, an error message if not */ public static function retrieve(string $url): Result { $doc = self::retrieveDocument($url) ->bind(fn(array $doc) => match ($doc['code']) { 200 => Result::OK($doc), default => Result::Error( "Prospective feed URL $url returned HTTP Code {$doc['code']}: {$doc['content']}"), }) ->bind(function (array $doc) use ($url) { $start = strtolower(strlen($doc['content']) >= 9 ? substr($doc['content'], 0, 9) : $doc['content']); return $start === 'bind(function (string $feedURL) use ($url) { if (!str_starts_with($feedURL, 'http')) { // Relative URL; feed should be retrieved in the context of the original URL $original = parse_url($url); $port = key_exists('port', $original) ? ":{$original['port']}" : ''; $feedURL = $original['scheme'] . '://' . $original['host'] . $port . $feedURL; } return self::retrieveDocument($feedURL); }) ->bind(fn($doc) => match ($doc['code']) { 200 => Result::OK($doc), default => Result::Error( "Derived feed URL {$doc['url']} returned HTTP Code {$doc['code']}: {$doc['content']}"), }) : Result::OK($doc); }); return $doc ->bind(fn($doc) => self::parseFeed($doc['content'])) ->bind(function (DOMDocument $parsed) use ($doc) { $extract = $parsed->getElementsByTagNameNS(self::ATOM_NS, 'feed')->length > 0 ? self::fromAtom(...) : self::fromRSS(...); return $extract($parsed, $doc->getOK()['url']); }); } }