<?php

namespace App\Services;

use App\DTOs\LandingPageAnalysis;
use App\Support\PublicHttp;
use DOMDocument;
use DOMXPath;
use Illuminate\Support\Collection;
use Illuminate\Support\Facades\Http;
use RuntimeException;
use Throwable;

/**
 * Reads a public landing page so the chat can suggest instead of interrogate.
 *
 * This is the first step of a campaign on every platform, which is why it lives
 * here rather than beside one platform's tools: the same page produces the
 * Meta objective, the Google keywords and the ad copy for both.
 *
 * Nothing here decides anything. It reports what the page says, and the caller
 * offers it as a suggestion the user can accept or overrule.
 */
class LandingPageAnalyzer
{
    private const MAX_HTML_BYTES = 2_000_000;

    private const MAX_TEXT_CHARS = 3000;

    /**
     * @throws RuntimeException when the page cannot or should not be read
     */
    public function analyze(string $url): LandingPageAnalysis
    {
        $host = parse_url($url, PHP_URL_HOST);

        if (! is_string($host) || $host === '') {
            throw new RuntimeException('That does not look like a web address. Check the URL and try again.');
        }

        // Both cases refuse the fetch, but they need different things from the
        // user: a typo needs correcting, a private address needs a public page.
        // Reported as one message each rather than one message for both.
        match ($this->reachability($host)) {
            'private' => throw new RuntimeException(
                'Use a public landing page URL. Local and private network addresses are not allowed.'
            ),
            'unresolvable' => throw new RuntimeException(
                "[{$host}] could not be found. Check the address is spelled correctly and is publicly reachable."
            ),
            default => null,
        };

        try {
            // Every hop checked, not just the one the user typed. The guard
            // above covers the first request; redirects are where a public
            // host hands the fetch to an internal one.
            $response = PublicHttp::send(
                Http::withHeaders(PublicHttp::browserHeaders())->connectTimeout(5)->timeout(15),
                $url,
            );
        } catch (Throwable $exception) {
            report($exception);

            throw new RuntimeException('The landing page could not be reached. Check the URL and try again.');
        }

        if ($response->failed()) {
            throw new RuntimeException('The landing page could not be read (HTTP '.$response->status().').');
        }

        $analysis = $this->parse($url, substr((string) $response->body(), 0, self::MAX_HTML_BYTES));

        // A page whose content only exists after its JavaScript runs is read
        // again in a browser.
        //
        // lucos.com answers 200 with 1362 bytes - a doctype, a CSP meta tag, a
        // favicon, a viewport and one script tag - so this produced a title and
        // nothing else, and every ad written from it said "Visit lucos.com.
        // Explore lucos.com to learn more." Four of Kaushal's five
        // conversations went that way, and the fault reads as bad copywriting
        // rather than as a page we could not see.
        //
        // Only when the first read came back empty, because rendering costs a
        // browser launch and several seconds. iblockads.org, attestpath.ai and
        // rebates.com are 13 to 43 times larger and never reach this.
        if ($this->looksClientRendered($analysis)) {
            $rendered = app(RenderedPageFetcher::class)->fetch($url);

            if (filled($rendered)) {
                $second = $this->parse($url, substr($rendered, 0, self::MAX_HTML_BYTES));

                // Kept only if it is actually better. A render that fails in
                // its own way - a consent wall, a bot check - can return less
                // than the plain fetch did, and replacing a thin analysis with
                // an empty one helps nobody.
                if (! $this->looksClientRendered($second)) {
                    return $second;
                }
            }
        }

        return $analysis;
    }

    /**
     * Whether the HTML we read carried no content of its own.
     *
     * A title with no description, no headings and almost no text is the
     * signature of an app shell. Any one of those alone is ordinary - plenty of
     * real pages omit a meta description - so all three have to hold.
     */
    private function looksClientRendered(LandingPageAnalysis $analysis): bool
    {
        return $analysis->headings === []
            && trim($analysis->description) === ''
            && mb_strlen(trim($analysis->text)) < 200;
    }

    private function parse(string $url, string $html): LandingPageAnalysis
    {
        $document = new DOMDocument;

        // A space before every tag, before parsing.
        //
        // textContent joins descendants with nothing in between, so a heading
        // marked up as <h1>Block<span>Annoying Ads</span></h1> reads back as
        // "BlockAnnoying Ads". That produced "blockannoying" as a keyword for
        // iblockads.org, which is not a word anybody searches for and would
        // have been bid on as a phrase match.
        //
        // Before every tag rather than only between adjacent ones: the fusing
        // boundary is text meeting a tag, which "><" does not match. Runs of
        // whitespace are collapsed below, so the extra spaces cost nothing.
        $html = str_replace('<', ' <', $html);

        // Real pages are rarely valid HTML, and a warning per malformed tag is
        // not news. The parser recovers on its own.
        @$document->loadHTML($html);
        $xpath = new DOMXPath($document);

        // Code is removed before any text is read out of the page.
        //
        // string(//body) returns every descendant's text, scripts and stylesheets
        // included, and an inline analytics snippet or a block of CSS is on almost
        // every landing page. So "function", "window", "datalayer", "gtag",
        // "webkit", "transition" and "important" were all being counted as things
        // the page is about - and GooglePublisher sends these to Google as PHRASE
        // match keywords on a live Search campaign, where they bid on traffic that
        // has nothing to do with the advertiser.
        //
        // The same text is also what the ad copy is written from, so this is not
        // only a keyword problem: the model was being shown JavaScript and asked
        // what the page sells.
        //
        // Removed from the document rather than filtered afterwards, because a list
        // of code words to exclude is a list that is always one framework behind.
        foreach (iterator_to_array($xpath->query('//script | //style | //noscript | //template') ?: []) as $node) {
            $node->parentNode?->removeChild($node);
        }

        $headings = collect($xpath->query('//h1 | //h2') ?: [])
            ->map(fn ($node): string => trim(preg_replace('/\s+/', ' ', $node->textContent)))
            ->filter()
            ->unique()
            ->take(8)
            ->values()
            ->all();

        $text = mb_substr(
            trim(preg_replace('/\s+/', ' ', (string) $xpath->evaluate('string(//body)'))),
            0,
            self::MAX_TEXT_CHARS,
        );

        return new LandingPageAnalysis(
            url: $url,
            title: trim((string) $xpath->evaluate('string(//title)')),
            description: trim((string) $xpath->evaluate('string(//meta[@name="description"]/@content)')),
            headings: $headings,
            text: $text,
            keywords: $this->keywords(
                $text,
                trim((string) $xpath->evaluate('string(//title)')),
                $headings,
            ),
        );
    }

    /**
     * English words that are never worth bidding on.
     *
     * Only the ones five characters or more, because shorter words are already
     * dropped. Not a general stop-word list: "free", "best" and "cheap" are
     * short or genuinely wanted in ad keywords, so this is confined to function
     * words and the furniture every website has.
     *
     * @var list<string>
     */
    private const NOT_KEYWORDS = [
        'about', 'above', 'after', 'again', 'against', 'along', 'already', 'although', 'always',
        'among', 'another', 'anything', 'around', 'because', 'become', 'been', 'before', 'being',
        'below', 'between', 'both', 'cannot', 'could', 'doing', 'during', 'each', 'either',
        'enough', 'especially', 'even', 'ever', 'every', 'everything', 'from', 'further', 'getting',
        'given', 'gives', 'going', 'have', 'having', 'here', 'however', 'into', 'itself', 'just',
        'keep', 'like', 'made', 'make', 'makes', 'many', 'might', 'more', 'most', 'much', 'must',
        'need', 'needs', 'never', 'often', 'once', 'only', 'other', 'others', 'ourselves', 'over',
        'own', 'perhaps', 'please', 'rather', 'really', 'same', 'should', 'simply', 'since',
        'some', 'something', 'still', 'such', 'take', 'takes', 'than', 'that', 'their', 'theirs',
        'them', 'themselves', 'then', 'there', 'therefore', 'these', 'they', 'thing', 'things',
        'this', 'those', 'through', 'today', 'together', 'under', 'until', 'upon', 'used', 'using',
        'very', 'want', 'wants', 'well', 'were', 'what', 'when', 'where', 'whether', 'which',
        'while', 'will', 'with', 'within', 'without', 'working', 'works', 'would', 'your',
        'yours', 'yourself',
        // The furniture, present on almost every page and meaningless as a bid.
        'click', 'clicks', 'contact', 'cookie', 'cookies', 'email', 'home', 'learn', 'menu',
        'page', 'pages', 'policy', 'privacy', 'read', 'reserved', 'rights', 'search', 'site',
        'submit', 'terms', 'thanks', 'welcome', 'website',
    ];

    /**
     * Keywords worth bidding on, from what the page is about.
     *
     * Was a frequency count of anything five characters or more, described in
     * this docblock as "a prompt for the user, not a keyword strategy". That
     * stopped being true: GooglePublisher sends these to Google as PHRASE match
     * keywords on a live Search campaign, so they are bid targets and nobody
     * sees them in between. iblockads.org produced "about", "works",
     * "experience", "online" and "smoother", every one of which would match a
     * great deal of traffic that has nothing to do with an ad blocker.
     *
     * Three changes. Function words and page furniture are dropped outright. A
     * word has to be repeated, which is what "the words the page repeats" meant
     * all along. And the title and headings are weighted, because that is where
     * a page says what it sells: an advertiser's brand usually appears once in
     * the body and cannot win on frequency.
     *
     * The repetition rule is relaxed rather than enforced when it would leave
     * too little: a short page can say something once and mean it, and an empty
     * list blocks the publish gate entirely.
     *
     * @param  list<string>  $headings
     * @return list<string>
     */
    private function keywords(string $text, string $title = '', array $headings = []): array
    {
        $words = fn (string $source): Collection => collect(preg_split('/[^\p{L}\p{N}]+/u', mb_strtolower($source)) ?: [])
            ->filter(fn (string $word): bool => mb_strlen($word) >= 5)
            ->reject(fn (string $word): bool => in_array($word, self::NOT_KEYWORDS, true))
            ->reject(fn (string $word): bool => ctype_digit($word));

        // The title and headings lead outright rather than competing on count.
        // "adblocker" appears once on iblockads.org, in the title, and is the
        // most valuable thing on the page to bid on; a frequency weighting left
        // it ranked below "smoother".
        $leading = $words($title.' '.implode(' ', $headings))->countBy()->sortDesc()->keys();

        $counted = $words($text)->countBy()->sortDesc();
        $repeated = $counted->filter(fn (int $count): bool => $count > 1)->keys();

        $chosen = $leading->merge($repeated)->unique();

        // A short page can say something once and mean it, and an empty list
        // blocks the publish gate outright, so the repetition rule gives way
        // rather than leaving a campaign with nothing to bid on.
        if ($chosen->count() < 5) {
            $chosen = $leading->merge($counted->keys())->unique();
        }

        return $chosen->take(10)->values()->all();
    }

    /**
     * Whether this host may be fetched, and if not, why not.
     *
     * A landing page URL arrives from whoever is chatting and the server will
     * fetch it, so without this the chat is a request-forgery tool pointed at
     * our own network. The name is resolved rather than pattern-matched,
     * because a perfectly public hostname can point at a private address.
     *
     * Anything that does not resolve is refused as well. That is deliberate: a
     * name we cannot resolve is a name we cannot vouch for.
     *
     * @return 'public'|'private'|'unresolvable'
     */
    private function reachability(string $host): string
    {
        $host = strtolower(rtrim($host, '.'));

        if ($host === 'localhost' || str_ends_with($host, '.local')) {
            return 'private';
        }

        $ip = gethostbyname($host);

        // gethostbyname hands back what it was given when it cannot resolve.
        if ($ip === $host && filter_var($host, FILTER_VALIDATE_IP) === false) {
            return 'unresolvable';
        }

        return filter_var($ip, FILTER_VALIDATE_IP, FILTER_FLAG_NO_PRIV_RANGE | FILTER_FLAG_NO_RES_RANGE) === false
            ? 'private'
            : 'public';
    }
}
