<?php

namespace App\Services;

use Illuminate\Http\Client\ConnectionException;
use Illuminate\Support\Str;

class SameHostPageFetcher
{
    public const MAX_FETCHES = 40;

    public const MAX_TEXT_CHARS = 12000;

    public const MAX_LINKS = 80;

    private int $remaining;

    /** @var list<string> */
    private array $disallows = [];

    private bool $robotsLoaded = false;

    public function __construct(
        private ProductPageScraper $scraper,
        private string $originUrl,
        int $maxFetches = self::MAX_FETCHES,
    ) {
        $this->remaining = $maxFetches;
    }

    public function remaining(): int
    {
        return $this->remaining;
    }

    public function origin(): string
    {
        $parts = parse_url($this->originUrl);
        $origin = ($parts['scheme'] ?? 'https').'://'.($parts['host'] ?? '');

        if (! empty($parts['port'])) {
            $origin .= ':'.$parts['port'];
        }

        return $origin;
    }

    public function isSameHost(string $url): bool
    {
        $host = $this->normalizeHost((string) parse_url($url, PHP_URL_HOST));

        return $host !== '' && $host === $this->originHost();
    }

    public function toAbsoluteUrl(string $url): ?string
    {
        $url = trim($url);

        if ($url === '' || str_starts_with($url, 'data:') || str_starts_with($url, 'javascript:') || str_starts_with($url, 'mailto:') || str_starts_with($url, 'tel:')) {
            return null;
        }

        if (str_starts_with($url, '#')) {
            return null;
        }

        if (str_starts_with($url, '//')) {
            $url = 'https:'.$url;
        }

        if (preg_match('#^https?://#i', $url) !== 1) {
            $origin = $this->origin();

            $url = str_starts_with($url, '/')
                ? $origin.$url
                : $origin.'/'.$url;
        }

        return $this->isSameHost($url) ? $url : null;
    }

    /**
     * Normalized form of a URL used for matching AI-returned URLs against
     * URLs we actually scraped (ignores scheme differences, trailing slash,
     * and www.).
     */
    public function normalizedUrl(string $url): string
    {
        $url = trim($url);

        if ($url === '') {
            return '';
        }

        $parts = parse_url($url);

        if ($parts === false) {
            return strtolower(rtrim($url, '/'));
        }

        $host = $this->normalizeHost((string) ($parts['host'] ?? ''));
        $path = rtrim((string) ($parts['path'] ?? ''), '/');
        $query = isset($parts['query']) ? '?'.$parts['query'] : '';

        return strtolower($host.$path.$query);
    }

    /**
     * @return array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked: bool}
     */
    public function fetch(string $url, string $type = ''): array
    {
        $this->loadRobots();

        if (! $this->isSameHost($url)) {
            return $this->skipped($url, 'host');
        }

        if ($this->remaining <= 0) {
            return $this->skipped($url, 'budget');
        }

        if ($this->isDisallowed($url)) {
            return $this->skipped($url, 'robots');
        }

        $this->remaining--;

        try {
            $result = $this->scraper->tryFetchHtml($url, ProductPageScraper::LISTING_TIMEOUT_SECONDS);
        } catch (ConnectionException) {
            return $this->skipped($url, 'timeout');
        }

        if ($result['timed_out']) {
            return $this->skipped($url, 'timeout');
        }

        $html = $result['html'];
        $blocked = $this->scraper->looksBlockedOrEmpty($html);

        return [
            'url' => $url,
            'text' => Str::limit($this->scraper->pageText($html), self::MAX_TEXT_CHARS, ''),
            'links' => $type == 'products' ? $this->extractProductsLinks($html) : $this->extractLinks($html),
            'skipped' => null,
            'remaining' => $this->remaining,
            'blocked' => $blocked,
        ];
    }

    /**
     * @return array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: string, remaining: int, blocked: bool}
     */
    private function skipped(string $url, string $reason): array
    {
        return [
            'url' => $url,
            'text' => '',
            'links' => [],
            'skipped' => $reason,
            'remaining' => $this->remaining,
            'blocked' => false,
        ];
    }

    /**
     * @return list<array{name: string, url: string}>
     */
    private function extractLinks(string $html): array
    {
        // Capture attrs before href, href itself, attrs after href, and inner HTML,
        // so we can fall back to aria-label/title/alt when there's no visible text
        // (e.g. icon-only or image-only nav links, which are common in ecommerce nav).
        if (preg_match_all('/<a\b((?:[^>]*?))href=["\']([^"\']+)["\']((?:[^>]*?))>(.*?)<\/a>/is', $html, $matches, PREG_SET_ORDER) === 0) {
            return [];
        }

        $links = [];
        $seen = [];

        foreach ($matches as $match) {
            $absolute = $this->toAbsoluteUrl(html_entity_decode($match[2], ENT_QUOTES | ENT_HTML5));
            $inner = trim(Str::of(strip_tags($match[4]))->squish()->toString());
            $name = $inner !== ''
                ? $inner
                : $this->fallbackLinkName($match[1].' '.$match[3], $match[4]);

            if ($absolute === null || $name === '' || isset($seen[$absolute]) || Str::length($name) > 60) {
                continue;
            }

            $seen[$absolute] = true;
            $links[] = [
                'name' => $name,
                'url' => $absolute,
            ];

            if (count($links) === self::MAX_LINKS) {
                break;
            }
        }

        return $links;
    }

    private function extractProductsLinks(string $html, array $productPatterns = []): array
    {
        /**
         * Extract:
         * - href attribute
         * - anchor HTML
         * - attributes before/after href
         */
        if (preg_match_all(
            '/<a\b([^>]*?)href=["\']([^"\']+)["\']([^>]*)>(.*?)<\/a>/is',
            $html,
            $matches,
            PREG_SET_ORDER
        ) === 0) {
            return [];
        }

        $products = [];
        $seen = [];

        /**
         * Default ecommerce product URL patterns.
         *
         * These can be overridden by passing $productPatterns.
         */
        if (empty($productPatterns)) {
            $productPatterns = [
                '/product/',
                '/products/',
                '/p/',
                '/a/',
                '/item/',
                '/dp/',
                '/buy/',
                '/shop/',
            ];
        }

        foreach ($matches as $match) {

            $href = html_entity_decode(
                trim($match[2]),
                ENT_QUOTES | ENT_HTML5
            );

            $absoluteUrl = $this->toAbsoluteUrl($href);

            if (! $absoluteUrl) {
                continue;
            }

            /**
             * Check whether this looks like a product URL.
             */
            $path = strtolower(
                parse_url($absoluteUrl, PHP_URL_PATH) ?? ''
            );

            $isProductUrl = false;

            foreach ($productPatterns as $pattern) {
                if (str_contains($path, strtolower($pattern))) {
                    $isProductUrl = true;
                    break;
                }
            }

            if (! $isProductUrl) {
                continue;
            }

            /**
             * Extract product name from anchor text.
             */
            $innerText = trim(
                Str::of(strip_tags($match[4]))
                    ->squish()
                    ->toString()
            );

            $name = $innerText !== ''
                ? $innerText
                : $this->fallbackLinkName(
                    $match[1].' '.$match[3],
                    $match[4]
                );

            if ($name === '') {
                continue;
            }

            /**
             * Remove duplicate URLs.
             */
            $key = rtrim($absoluteUrl, '/');

            if (isset($seen[$key])) {
                continue;
            }

            $seen[$key] = true;

            $products[] = [
                'name' => $name,
                'url' => $absoluteUrl,
            ];
        }

        return $products;
    }

    /**
     * Derive a name for a link with no visible text: image-only or icon-only
     * anchors (common in ecommerce category nav). Checks the anchor's own
     * aria-label/title first, then the inner image's alt text.
     */
    private function fallbackLinkName(string $anchorAttrs, string $inner): string
    {
        if (preg_match('/aria-label=["\']([^"\']+)["\']/i', $anchorAttrs, $m) === 1) {
            return trim(html_entity_decode($m[1], ENT_QUOTES | ENT_HTML5));
        }

        if (preg_match('/title=["\']([^"\']+)["\']/i', $anchorAttrs, $m) === 1) {
            return trim(html_entity_decode($m[1], ENT_QUOTES | ENT_HTML5));
        }

        if (preg_match('/alt=["\']([^"\']+)["\']/i', $inner, $m) === 1) {
            return trim(html_entity_decode($m[1], ENT_QUOTES | ENT_HTML5));
        }

        return '';
    }

    private function originHost(): string
    {
        return $this->normalizeHost((string) parse_url($this->originUrl, PHP_URL_HOST));
    }

    private function normalizeHost(string $host): string
    {
        $host = strtolower($host);

        return str_starts_with($host, 'www.') ? substr($host, 4) : $host;
    }

    private function loadRobots(): void
    {
        if ($this->robotsLoaded) {
            return;
        }

        $this->robotsLoaded = true;

        if ($this->remaining <= 0) {
            return;
        }

        $this->remaining--;
        $body = $this->scraper->fetchHtml($this->origin().'/robots.txt');
        $this->disallows = $this->parseRobots($body);
    }

    /**
     * @return list<string>
     */
    private function parseRobots(string $body): array
    {
        $disallows = [];
        $applies = false;

        foreach (preg_split('/\r\n|\r|\n/', $body) ?: [] as $line) {
            $line = trim($line);

            if ($line === '' || str_starts_with($line, '#')) {
                continue;
            }

            if (preg_match('/^user-agent:\s*(.+)$/i', $line, $matches) === 1) {
                $applies = trim($matches[1]) === '*';

                continue;
            }

            if ($applies && preg_match('/^disallow:\s*(.*)$/i', $line, $matches) === 1) {
                $path = trim($matches[1]);

                if ($path !== '') {
                    $disallows[] = $path;
                }
            }
        }

        return $disallows;
    }

    private function isDisallowed(string $url): bool
    {
        if ($this->isSubmittedUrl($url)) {
            return false;
        }

        $path = (string) (parse_url($url, PHP_URL_PATH) ?: '/');

        if ($this->isAlwaysAllowed($path)) {
            return false;
        }

        foreach ($this->disallows as $rule) {
            if ($rule === '/') {
                continue;
            }

            if (str_starts_with($path, $rule)) {
                return true;
            }
        }

        return false;
    }

    private function isSubmittedUrl(string $url): bool
    {
        $submitted = parse_url($this->originUrl);
        $candidate = parse_url($url);

        return $this->normalizeHost((string) ($submitted['host'] ?? '')) === $this->normalizeHost((string) ($candidate['host'] ?? ''))
            && ($submitted['path'] ?? '/') === ($candidate['path'] ?? '/');
    }

    private function isAlwaysAllowed(string $path): bool
    {
        $file = strtolower((string) basename($path));

        return $path === '/robots.txt'
            || $file === 'robots.txt'
            || $file === 'sitemap.xml'
            || str_ends_with($file, 'sitemap.xml');
    }
}
