<?php

namespace App\Services\AI;

use App\Contracts\ListingScraper;
use App\DTOs\ScrapedListingProduct;
use App\Services\ProductPageScraper;
use App\Services\SameHostPageFetcher;
use Illuminate\Contracts\JsonSchema\JsonSchema;
use Illuminate\Support\Facades\Log;
use Illuminate\Support\Str;

use function Laravel\Ai\agent;

class CatalogWebsiteResearch
{
    private const PUPPETEER_MIN_PRODUCTS = 5;

    private ?string $lastListingScrapeError = null;

    public function __construct(
        private ProductPageScraper $scraper,
        private ListingScraper $listingScraper,
    ) {}

    public function lastListingScrapeError(): ?string
    {
        return $this->lastListingScrapeError;
    }

    /**
     * @return list<array{name: string, url: string}>
     */
    public function categories(string $websiteUrl): array
    {
        $fetcher = new SameHostPageFetcher($this->scraper, $websiteUrl);
        $home = $fetcher->fetch($websiteUrl);
        $sitemap = $fetcher->fetch($fetcher->origin().'/sitemap.xml');

        Log::info('Catalog category URL fetch', [
            'website' => $websiteUrl,
            'home' => $this->pageFetchLog($home),
            'sitemap' => $this->pageFetchLog($sitemap),
        ]);

        if ($home['blocked'] ?? false) {
            Log::warning('Catalog research: homepage looks blocked or empty', [
                'website' => $websiteUrl,
                'text_len' => strlen($home['text']),
            ]);
        }

        $knownLinks = [...$home['links'], ...$sitemap['links']];

        $prompt = "Target website: {$websiteUrl}\nFetch budget remaining: {$fetcher->remaining()}\n\n"
            ."Homepage text:\n{$this->pageBlock($home)}\n\n"
            ."Sitemap text:\n{$this->pageBlock($sitemap)}\n\n"
            .'List the site\'s main product categories from navigation and the extracted links. '
            .'The url field must be copied exactly from an entry under "Extracted links" above. '
            .'If no matching link exists for a category, return an empty string for url — never construct, modify, or guess a URL. '
            .'If you cannot find categories, return an empty list.';

        $response = agent(
            instructions: 'You map ecommerce sites into shoppable categories. Prefer names from navigation and extracted links. Every url you return must be copied verbatim from the Extracted links list you were given. Never invent, guess, or modify a URL.',
            schema: fn (JsonSchema $schema) => [
                'categories' => $schema->array()->items(
                    $schema->object([
                        'name' => $schema->string()->description('Category name as shown on the site.')->required(),
                        'url' => $schema->string()->description('Same-host category URL copied verbatim from Extracted links, or empty.')->required(),
                    ])
                )->description('Main product categories.')->required(),
            ],
        )->prompt($prompt, provider: TextProviderFailover::providers());

        $raw = $response->toArray();
        $categories = $this->namedUrls($this->itemsFrom($raw, 'categories'), $fetcher, $knownLinks);

        if ($categories === []) {
            $categories = $this->fallbackNamedLinks($home['links']);
        }

        Log::info('Catalog categories AI response', [
            'website' => $websiteUrl,
            'home_skipped' => $home['skipped'],
            'home_blocked' => $home['blocked'] ?? false,
            'home_text_len' => strlen($home['text']),
            'home_links_count' => count($home['links']),
            'sitemap_links_count' => count($sitemap['links']),
            'raw_response' => $raw,
            'resolved_categories' => $categories,
        ]);

        return $categories;
    }

    /**
     * @return list<array{name: string, url: string}>
     */
    public function subcategories(string $websiteUrl, string $categoryName, ?string $categoryUrl = null): array
    {
        $fetcher = new SameHostPageFetcher($this->scraper, $websiteUrl);
        $pageUrl = filled($categoryUrl) ? $categoryUrl : $websiteUrl;
        $page = $fetcher->fetch($pageUrl);

        Log::info('Catalog subcategory URL fetch', [
            'website' => $websiteUrl,
            'category' => $categoryName,
            'page' => $this->pageFetchLog($page),
        ]);

        if ($page['blocked'] ?? false) {
            Log::warning('Catalog research: category page looks blocked or empty', [
                'website' => $websiteUrl,
                'category_url' => $pageUrl,
                'text_len' => strlen($page['text']),
            ]);
        }

        $prompt = "Target website: {$websiteUrl}\nSelected category: {$categoryName}\nCategory URL: {$pageUrl}\nFetch budget remaining: {$fetcher->remaining()}\n\n"
            ."Page text:\n{$this->pageBlock($page)}\n\n"
            .'List nested product departments under this category (for example Tents under Outdoor). '
            .'The url field must be copied exactly from an entry under "Extracted links" above. '
            .'If no matching link exists for a subcategory, return an empty string for url — never construct, modify, or guess a URL. '
            .'Do not list products, product cards, "View …" links, account pages, membership, quizzes, or site chrome. '
            .'If this page is already a product listing or has no nested departments, return an empty list.';

        $response = agent(
            instructions: 'You list nested product departments for one category on an ecommerce site. Return only department links, never products or chrome. If the page is already a product listing, return an empty list. Every url you return must be copied verbatim from the Extracted links list you were given. Never invent, guess, or modify a URL.',
            schema: fn (JsonSchema $schema) => [
                'subcategories' => $schema->array()->items(
                    $schema->object([
                        'name' => $schema->string()->description('Subcategory name as shown on the site.')->required(),
                        'url' => $schema->string()->description('Same-host subcategory URL copied verbatim from Extracted links, or empty.')->required(),
                    ])
                )->description('Nested departments of the selected category. Empty when the category is already a product listing.')->required(),
            ],
        )->prompt($prompt, provider: TextProviderFailover::providers());

        $raw = $response->toArray();
        $subcategories = $this->departmentChoices(
            $this->namedUrls($this->itemsFrom($raw, 'subcategories'), $fetcher, $page['links']),
        );

        Log::info('Catalog subcategories AI response', [
            'website' => $websiteUrl,
            'category' => $categoryName,
            'page_skipped' => $page['skipped'],
            'page_blocked' => $page['blocked'] ?? false,
            'page_text_len' => strlen($page['text']),
            'page_links_count' => count($page['links']),
            'raw_response' => $raw,
            'resolved_subcategories' => $subcategories,
        ]);

        return $subcategories;
    }

    /**
     * @param  list<string>  $listingUrls
     * @return list<array{name: string, url: string, popularity_note: string}>
     */
    public function topProducts(
        string $websiteUrl,
        string $categoryName = '',
        string $subcategoryName = '',
        array $listingUrls = [],
    ): array {
        if (config('scraping.puppeteer.enabled')) {
            $puppeteerProducts = $this->topProductsFromListingScraper($listingUrls);

            if ($puppeteerProducts !== null) {
                return $puppeteerProducts;
            }

            Log::info('Catalog product discovery falling back to HTTP research', [
                'website' => $websiteUrl,
                'listing_urls' => $listingUrls,
            ]);
        }

        $fetcher = new SameHostPageFetcher($this->scraper, $websiteUrl);
        $pages = $this->listingPages($fetcher, $websiteUrl, $listingUrls, 'products');
        $resolvedListingUrls = array_column($pages, 'url');
        $scope = $this->productScope($categoryName, $subcategoryName, $resolvedListingUrls);
        $pages = $this->constrainListingPages($pages, $resolvedListingUrls);

        Log::info('Catalog listing URL fetch', [
            'website' => $websiteUrl,
            'category' => $categoryName,
            'subcategory' => $subcategoryName,
            'pages' => $pages,
        ]);

        $knownLinks = collect($pages)->flatMap(fn (array $page): array => $page['links'])->all();

        $anyBlocked = collect($pages)->contains(fn (array $page): bool => $page['blocked'] ?? false);

        if ($anyBlocked) {
            Log::warning('Catalog research: one or more listing pages look blocked or empty', [
                'website' => $websiteUrl,
                'listing_urls' => $listingUrls,
            ]);
        }

        $listings = collect($pages)
            ->map(function (array $page): string {
                $links = collect($page['links'] ?? [])
                    ->map(fn (array $link): string => "- {$link['name']}: {$link['url']}")
                    ->implode("\n");

                if ($links === '') {
                    $links = '(no product links extracted)';
                }

                return "Listing URL: {$page['url']}\n\nExtracted Product Links (listing/grid order):\n".$links;
            })
            ->implode("\n\n");

        $prompt = "Target website: {$websiteUrl}\n{$scope}\nFetch budget remaining: {$fetcher->remaining()}\n\n"
            ."{$listings}\n\n"
            .'Pick products that appear on those exact listing page(s). Prefer listing/grid order first. '
            .'Among those same extracted links, you may prefer items marked top-rated, bestseller, or featured. '
            .'Do not pick bestsellers from elsewhere on the site, and do not expand beyond the extracted product links. '
            .'Return up to 12 candidates. The url field must be copied exactly from Extracted Product Links — never construct, modify, or guess a URL. '
            .'Never return category or search hub URLs. If the listing has 5 or more extracted product links, return at least 5 of them. '
            .'Only return fewer than 5 when fewer than 5 extracted product links exist.';

        return $this->pickProductsFromAgent(
            $prompt,
            'You pick products from one or more specific ecommerce listing pages. '
            .'The Extracted Product Links are already the product URLs from those pages, in listing/grid order. '
            .'You MUST select products only from those links. Prefer that listing order, then ratings among those links. '
            .'If the listing URL has no query string, do not treat a category name as a filter — use the extracted links as they are. '
            .'Do not pick products from the rest of the website. If 5 or more extracted product links exist, return at least 5. '
            .'Every url must be copied exactly. Never create or modify URLs.',
            $fetcher,
            $websiteUrl,
            $categoryName,
            $subcategoryName,
            $listingUrls,
            $knownLinks,
            requireKnownUrl: true,
            anyListingBlocked: $anyBlocked,
            fallback: false,
        );
    }

    /**
     * @param  list<array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked: bool}>  $pages
     */
    private function listingNeedsAiFallback(array $pages): bool
    {
        return collect($pages)->every(
            fn (array $page): bool => $page['skipped'] !== null || $page['links'] === [],
        );
    }

    /**
     * @param  list<string>  $listingUrls
     * @param  list<array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked: bool}>  $pages
     * @return list<array{name: string, url: string, popularity_note: string}>
     */
    private function topProductsFromCategoryLinks(
        SameHostPageFetcher $fetcher,
        string $websiteUrl,
        string $categoryName,
        string $subcategoryName,
        array $listingUrls,
        array $pages,
    ): array {
        $links = collect($listingUrls)
            ->map(fn (string $url): string => trim($url))
            ->filter()
            ->unique()
            ->values();

        if ($links->isEmpty()) {
            $links = collect([$websiteUrl]);
        }

        $scope = $this->productScope($categoryName, $subcategoryName, $links->all());

        $skipReasons = collect($pages)
            ->map(fn (array $page): string => $page['url'].' ('.($page['skipped'] ?? 'empty').')')
            ->implode("\n");

        Log::warning('Catalog listing fetch failed; asking AI for top products', [
            'website' => $websiteUrl,
            'category' => $categoryName,
            'subcategory' => $subcategoryName,
            'listing_urls' => $links->all(),
            'pages' => array_map(fn (array $page): array => $this->pageFetchLog($page), $pages),
        ]);

        $prompt = "Target website: {$websiteUrl}\n{$scope}\n\n"
            ."The live listing pages could not be scraped:\n{$skipReasons}\n\n"
            ."Use these category or subcategory URLs as the department to search:\n"
            .$links->map(fn (string $url): string => "- {$url}")->implode("\n")."\n\n"
            .'Return up to 12 real product page URLs on this host for the top-rated, bestselling, or highest-search-volume products in that department. '
            .'URLs must be product pages on this host, never category, search, or filter hubs. '
            .'If you cannot name real products sold on this site, return an empty list. Never invent URLs on other hosts.';

        return $this->pickProductsFromAgent(
            $prompt,
            'You pick top-rated, bestselling, or highest-search-volume products for one retailer department when the listing page cannot be fetched. Return real same-host product page URLs for that category or subcategory. Never invent URLs on other hosts. Never pad a short list.',
            $fetcher,
            $websiteUrl,
            $categoryName,
            $subcategoryName,
            $links->all(),
            [],
            requireKnownUrl: false,
            anyListingBlocked: true,
            fallback: true,
        );
    }

    /**
     * @param  list<string>  $listingUrls
     * @param  list<array{name: string, url: string}>  $knownLinks
     * @return list<array{name: string, url: string, popularity_note: string}>
     */
    private function pickProductsFromAgent(
        string $prompt,
        string $instructions,
        SameHostPageFetcher $fetcher,
        string $websiteUrl,
        string $categoryName,
        string $subcategoryName,
        array $listingUrls,
        array $knownLinks,
        bool $requireKnownUrl,
        bool $anyListingBlocked,
        bool $fallback,
    ): array {
        $response = agent(
            instructions: $instructions,
            schema: fn (JsonSchema $schema) => [
                'products' => $schema->array()->items(
                    $schema->object([
                        'name' => $schema->string()->description('Product name as sold.')->required(),
                        'url' => $schema->string()->description('Same-host product page URL.')->required(),
                        'popularity_note' => $schema->string()->description('Why this is a top pick, e.g. rating, bestseller, or search volume.')->required(),
                    ])
                )->description('Top-rated or most popular product candidates, up to 12.')->required(),
            ],
        )->prompt($prompt, provider: TextProviderFailover::providers());

        $raw = $response->toArray();
        $candidates = $this->padCandidatesFromListing($this->itemsFrom($raw, 'products'), $knownLinks);
        $products = [];
        $dropped = [];

        foreach ($candidates as $row) {
            if (! is_array($row)) {
                continue;
            }

            $name = trim((string) ($row['name'] ?? ''));
            $url = $requireKnownUrl
                ? $this->resolvedKnownUrl((string) ($row['url'] ?? ''), $fetcher, $knownLinks)
                : $this->resolvedUrl((string) ($row['url'] ?? ''), $fetcher);
            $note = trim((string) ($row['popularity_note'] ?? ''));

            if ($name === '' || $url === '') {
                $dropped[] = ['reason' => 'missing_name_or_unmatched_url', 'row' => $row];

                continue;
            }

            if (! $this->productExistsWithImage($url)) {
                $dropped[] = ['reason' => 'no_live_page_or_image', 'url' => $url];

                continue;
            }

            $products[] = [
                'name' => $name,
                'url' => $url,
                'popularity_note' => $note,
            ];

            if (count($products) === 6) {
                break;
            }
        }

        Log::info($fallback ? 'Catalog top products AI fallback response' : 'Catalog top products AI response', [
            'website' => $websiteUrl,
            'category' => $categoryName,
            'subcategory' => $subcategoryName,
            'listing_urls' => $listingUrls,
            'any_listing_blocked' => $anyListingBlocked,
            'fallback' => $fallback,
            'candidate_count' => is_countable($candidates) ? count($candidates) : 0,
            'accepted_count' => count($products),
            'dropped' => $dropped,
            'raw_response' => $raw,
        ]);

        return $products;
    }

    /**
     * @param  list<array{name: string, url: string}>  $knownLinks
     * @return list<array<string, mixed>>
     */
    private function padCandidatesFromListing(mixed $candidates, array $knownLinks): array
    {
        $rows = [];
        $seen = [];

        foreach (is_array($candidates) ? $candidates : [] as $row) {
            if (! is_array($row)) {
                continue;
            }

            $rows[] = $row;
            $url = Str::lower(trim((string) ($row['url'] ?? '')));

            if ($url !== '') {
                $seen[$url] = true;
            }
        }

        if (count($rows) >= 5 || $knownLinks === []) {
            return $rows;
        }

        Log::warning(
            $rows === []
                ? 'AI returned empty products, using extracted links fallback'
                : 'AI returned fewer than 5 products, filling from extracted links',
        );

        foreach ($knownLinks as $link) {
            $url = trim((string) ($link['url'] ?? ''));
            $key = Str::lower($url);

            if ($url === '' || isset($seen[$key])) {
                continue;
            }

            $seen[$key] = true;
            $rows[] = [
                'name' => trim((string) ($link['name'] ?? '')),
                'url' => $url,
                'popularity_note' => 'Extracted from product listing page',
            ];

            if (count($rows) === 12) {
                break;
            }
        }

        return $rows;
    }

    /**
     * @param  list<string>  $listingUrls
     * @return list<array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked: bool}>
     */
    private function listingPages(SameHostPageFetcher $fetcher, string $websiteUrl, array $listingUrls, string $type = ''): array
    {
        $urls = [];

        foreach ($listingUrls as $url) {
            $url = trim($url);

            if ($url !== '' && ! in_array($url, $urls, true)) {
                $urls[] = $url;
            }
        }

        if ($urls === []) {
            $urls[] = $websiteUrl;
        }

        return array_map(fn (string $url): array => $fetcher->fetch($url, $type), $urls);
    }

    /**
     * @param  list<string>  $listingUrls
     */
    private function productScope(string $categoryName, string $subcategoryName, array $listingUrls = []): string
    {
        $listingBlock = $this->listingScope($listingUrls);

        if (! $this->listingHasQueryFilter($listingUrls) && $listingBlock !== '') {
            return "This listing URL has no query-string filter. Pick products from every extracted link on the listing page(s) below. Do not treat a category or subcategory name as a product filter.\n{$listingBlock}";
        }

        if ($categoryName !== '' && $subcategoryName !== '') {
            return "Category: {$categoryName}\nSubcategories: {$subcategoryName}\n{$listingBlock}";
        }

        if ($categoryName !== '') {
            return "Category: {$categoryName}\nNo subcategory was found. Use only the listing page(s) below.\n{$listingBlock}";
        }

        if ($listingBlock !== '') {
            return "No category or subcategory was selected. Use only these listing page(s) — do not pick products from the rest of the site.\n{$listingBlock}";
        }

        return 'No listing URL was provided. Return an empty list rather than guessing site-wide products.';
    }

    /**
     * @param  list<string>  $listingUrls
     */
    private function listingHasQueryFilter(array $listingUrls): bool
    {
        return $this->listingFilterTokens($listingUrls) !== [];
    }

    /**
     * @param  list<string>  $listingUrls
     */
    private function listingScope(array $listingUrls): string
    {
        $urls = collect($listingUrls)
            ->map(fn (string $url): string => trim($url))
            ->filter()
            ->unique()
            ->values();

        if ($urls->isEmpty()) {
            return '';
        }

        return "Listing URL(s) — pick products that appear on these exact pages, including any query string:\n"
            .$urls->map(fn (string $url): string => "- {$url}")->implode("\n");
    }

    /**
     * @param  list<array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked: bool}>  $pages
     * @param  list<string>  $listingUrls
     * @return list<array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked: bool}>
     */
    private function constrainListingPages(array $pages, array $listingUrls): array
    {
        $tokens = $this->listingFilterTokens($listingUrls);
        $allLinks = collect($pages)->flatMap(fn (array $page): array => $page['links'])->all();
        $knownLinks = $this->constrainProductLinks($allLinks, $tokens);

        if ($knownLinks === $allLinks) {
            return $pages;
        }

        $allowedUrls = array_flip(array_column($knownLinks, 'url'));

        return array_map(function (array $page) use ($allowedUrls): array {
            $page['links'] = array_values(array_filter(
                $page['links'],
                fn (array $link): bool => isset($allowedUrls[$link['url']]),
            ));

            return $page;
        }, $pages);
    }

    /**
     * @param  list<string>  $listingUrls
     * @return list<string>
     */
    private function listingFilterTokens(array $listingUrls): array
    {
        $tokens = [];

        foreach ($listingUrls as $url) {
            $parts = parse_url($url);

            if ($parts === false) {
                continue;
            }

            $query = [];
            parse_str((string) ($parts['query'] ?? ''), $query);

            foreach ($query as $key => $value) {
                if ($this->isIgnoredListingQueryKey((string) $key)) {
                    continue;
                }

                foreach ((array) $value as $part) {
                    foreach ($this->listingTokenWords((string) $part) as $word) {
                        $tokens[] = $word;
                    }
                }
            }
        }

        return array_values(array_slice(array_unique($tokens), 0, 8));
    }

    /**
     * @return list<string>
     */
    private function listingTokenWords(string $value): array
    {
        $words = preg_split('/[^a-z0-9]+/i', Str::lower($value)) ?: [];
        $tokens = [];

        foreach ($words as $word) {
            $word = trim($word);

            if (Str::length($word) < 4 || in_array($word, $this->ignoredListingWords(), true)) {
                continue;
            }

            if (str_ends_with($word, 's') && Str::length($word) >= 5) {
                $word = substr($word, 0, -1);
            }

            $tokens[] = $word;
        }

        return $tokens;
    }

    /**
     * @param  list<array{name: string, url: string}>  $links
     * @param  list<string>  $tokens
     * @return list<array{name: string, url: string}>
     */
    private function constrainProductLinks(array $links, array $tokens): array
    {
        if ($tokens === [] || $links === []) {
            return $links;
        }

        $matched = array_values(array_filter(
            $links,
            fn (array $link): bool => $this->linkMatchesAllListingTokens($link, $tokens),
        ));

        return count($matched) >= 5 ? $matched : $links;
    }

    /**
     * @param  array{name: string, url: string}  $link
     * @param  list<string>  $tokens
     */
    private function linkMatchesAllListingTokens(array $link, array $tokens): bool
    {
        $haystack = Str::lower(trim($link['name'] ?? '').' '.trim($link['url'] ?? ''));

        foreach ($tokens as $token) {
            if (preg_match('/(?:^|[^a-z0-9])'.preg_quote($token, '/').'(?:[^a-z0-9]|$)/', $haystack) !== 1) {
                return false;
            }
        }

        return true;
    }

    private function isIgnoredListingQueryKey(string $key): bool
    {
        $key = Str::lower($key);

        if (str_starts_with($key, 'utm_')) {
            return true;
        }

        return in_array($key, [
            'page',
            'pagenum',
            'p',
            'offset',
            'limit',
            'per_page',
            'perpage',
            'sort',
            'sort_by',
            'sortby',
            'order',
            'orderby',
            'view',
            'display',
            'cursor',
            'before',
            'after',
            'ref',
            'fbclid',
            'gclid',
            'gbraid',
            'wbraid',
            'msclkid',
        ], true);
    }

    /**
     * @return list<string>
     */
    private function ignoredListingWords(): array
    {
        return [
            'true',
            'false',
            'with',
            'from',
            'this',
            'that',
            'html',
            'utf8',
            'and',
            'the',
        ];
    }

    private function productExistsWithImage(string $url): bool
    {
        $html = $this->scraper->fetchLiveHtml($url);
        $hasImage = $html !== null && filled($this->scraper->primaryImageUrl($html, $url));

        Log::info('Catalog product URL fetch', [
            'url' => $url,
            'live' => $html !== null,
            'has_image' => $hasImage,
        ]);

        return $hasImage;
    }

    /**
     * @param  array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked?: bool}  $page
     * @return array{url: string, skipped: ?string, blocked: bool, links_count: int, remaining: int}
     */
    private function pageFetchLog(array $page): array
    {
        return [
            'url' => $page['url'],
            'skipped' => $page['skipped'],
            'blocked' => (bool) ($page['blocked'] ?? false),
            'links_count' => count($page['links']),
            'remaining' => $page['remaining'],
        ];
    }

    /**
     * @param  array<string, mixed>  $data
     */
    private function itemsFrom(array $data, string $key): mixed
    {
        return $data[$key] ?? $data[Str::ucfirst($key)] ?? [];
    }

    /**
     * Resolve AI-returned name/url pairs, validating each URL against a set
     * of URLs we actually scraped from the page(s). This prevents the model
     * from returning plausible-looking but hallucinated or malformed URLs.
     *
     * @param  list<array{name: string, url: string}>  $knownLinks
     * @return list<array{name: string, url: string}>
     */
    private function namedUrls(mixed $items, SameHostPageFetcher $fetcher, array $knownLinks = []): array
    {
        $known = $this->indexKnownLinks($fetcher, $knownLinks);
        $named = [];

        foreach (is_array($items) ? $items : [] as $row) {
            if (is_string($row)) {
                $name = trim($row);
                $url = '';
            } elseif (is_array($row)) {
                $name = trim((string) ($row['name'] ?? ''));
                $url = $this->matchKnownUrl((string) ($row['url'] ?? ''), $name, $fetcher, $known);
            } else {
                continue;
            }

            if ($name === '') {
                continue;
            }

            $named[] = [
                'name' => $name,
                'url' => $url,
            ];
        }

        return $named;
    }

    /**
     * Resolve a single AI-returned product URL against known links, dropping
     * it (returning '') if it doesn't match anything we actually scraped.
     *
     * @param  list<array{name: string, url: string}>  $knownLinks
     */
    private function resolvedKnownUrl(string $url, SameHostPageFetcher $fetcher, array $knownLinks): string
    {
        $known = $this->indexKnownLinks($fetcher, $knownLinks);
        $normalized = $fetcher->normalizedUrl($this->resolvedUrl($url, $fetcher));

        return $known['by_url'][$normalized] ?? '';
    }

    /**
     * @param  list<array{name: string, url: string}>  $knownLinks
     * @return array{by_url: array<string, string>, by_name: array<string, string>}
     */
    private function indexKnownLinks(SameHostPageFetcher $fetcher, array $knownLinks): array
    {
        $byUrl = [];
        $byName = [];

        foreach ($knownLinks as $link) {
            $url = trim($link['url'] ?? '');
            $name = trim($link['name'] ?? '');

            if ($url === '') {
                continue;
            }

            $byUrl[$fetcher->normalizedUrl($url)] = $url;

            if ($name !== '') {
                $byName[Str::lower($name)] = $url;
            }
        }

        return ['by_url' => $byUrl, 'by_name' => $byName];
    }

    /**
     * @param  array{by_url: array<string, string>, by_name: array<string, string>}  $known
     */
    private function matchKnownUrl(string $aiUrl, string $name, SameHostPageFetcher $fetcher, array $known): string
    {
        if ($known['by_url'] === []) {
            // No known links to validate against (e.g. fallback path) — accept
            // a resolved same-host URL as-is rather than dropping everything.
            return $this->resolvedUrl($aiUrl, $fetcher);
        }

        $resolved = $this->resolvedUrl($aiUrl, $fetcher);
        $normalized = $fetcher->normalizedUrl($resolved);

        if ($normalized !== '' && isset($known['by_url'][$normalized])) {
            return $known['by_url'][$normalized];
        }

        // The AI's URL didn't match a scraped link exactly (e.g. paraphrased
        // or truncated it). Fall back to matching by the exact link name
        // rather than trusting the possibly-hallucinated URL.
        return $known['by_name'][Str::lower($name)] ?? '';
    }

    /**
     * @param  list<array{name: string, url: string}>  $links
     * @return list<array{name: string, url: string}>
     */
    private function fallbackNamedLinks(array $links): array
    {
        $named = [];
        $seen = [];

        foreach ($links as $link) {
            $name = trim($link['name'] ?? '');
            $url = trim($link['url'] ?? '');

            if ($name === '' || $url === '' || isset($seen[Str::lower($name)]) || $this->isUtilityLink($name, $url)) {
                continue;
            }

            $seen[Str::lower($name)] = true;
            $named[] = [
                'name' => $name,
                'url' => $url,
            ];

            if (count($named) === 20) {
                break;
            }
        }

        return $named;
    }

    private function resolvedUrl(string $url, SameHostPageFetcher $fetcher): string
    {
        $url = trim($url);

        if ($url === '') {
            return '';
        }

        $absolute = $fetcher->toAbsoluteUrl($url);

        return $absolute ?? '';
    }

    private function isUtilityLink(string $name, string $url): bool
    {
        $haystack = Str::lower($name.' '.$url);

        return (bool) preg_match(
            '/\b(login|log in|sign in|sign up|account|cart|checkout|wishlist|help|faq|privacy|terms|cookie|gift card|store locator|email|instagram|facebook|twitter|youtube|pinterest|membership|subscribe|subscription|order history|quiz)\b/i',
            $haystack,
        );
    }

    /**
     * @param  list<array{name: string, url: string}>  $rows
     * @return list<array{name: string, url: string}>
     */
    private function departmentChoices(array $rows): array
    {
        $departments = [];

        foreach ($rows as $row) {
            $name = trim((string) ($row['name'] ?? ''));
            $url = trim((string) ($row['url'] ?? ''));

            if ($name === '' || $this->isNonDepartmentLink($name, $url)) {
                continue;
            }

            $departments[] = [
                'name' => $name,
                'url' => $url,
            ];
        }

        return $departments;
    }

    private function isNonDepartmentLink(string $name, string $url): bool
    {
        if ($this->isUtilityLink($name, $url)) {
            return true;
        }

        return preg_match('/^view\s+/i', $name) === 1;
    }

    /**
     * @param  array{url: string, text: string, links: list<array{name: string, url: string}>, skipped: ?string, remaining: int, blocked?: bool}  $page
     */
    private function pageBlock(array $page): string
    {
        if ($page['skipped'] !== null) {
            return "(skipped: {$page['skipped']})";
        }

        if ($page['blocked'] ?? false) {
            return "(warning: this page returned very little content or looks like a bot-protection/JavaScript-only page — treat any extracted links/text below as unreliable)\n"
                .($page['text'] !== '' ? $page['text'] : '(empty page)');
        }

        $text = $page['text'] !== '' ? $page['text'] : '(empty page)';
        $links = $page['links'] === []
            ? '(no links extracted)'
            : collect($page['links'])
                ->take(40)
                ->map(fn (array $link): string => "{$link['name']} | {$link['url']}")
                ->implode("\n");

        return $text."\n\nExtracted links:\n".$links;
    }

    /**
     * @param  list<string>  $listingUrls
     * @return list<array<string, mixed>>|null
     */
    private function topProductsFromListingScraper(array $listingUrls): ?array
    {
        $this->lastListingScrapeError = null;

        $urls = collect($listingUrls)
            ->map(fn (mixed $url): string => trim((string) $url))
            ->filter()
            ->unique()
            ->values();

        if ($urls->isEmpty()) {
            return null;
        }

        $products = [];
        $seenUrls = [];
        $errors = [];

        foreach ($urls as $listingUrl) {
            $result = $this->listingScraper->scrape($listingUrl);

            if (! $result->succeeded()) {
                $error = trim((string) $result->error);

                if ($error !== '') {
                    $errors[] = $error;
                }

                Log::warning('Puppeteer listing scrape failed for catalog discovery', [
                    'listing_url' => $listingUrl,
                    'error' => $result->error,
                ]);

                continue;
            }

            foreach ($result->products as $product) {
                $key = Str::lower(rtrim($product->url, '/'));

                if ($key === '' || isset($seenUrls[$key])) {
                    continue;
                }

                $seenUrls[$key] = true;
                $products[] = $product;
            }
        }

        if (count($products) < self::PUPPETEER_MIN_PRODUCTS) {
            $this->lastListingScrapeError = $errors[0]
                ?? (count($products) === 0
                    ? 'No products extracted from the listing page.'
                    : 'Too few products extracted from the listing page.');

            Log::info('Puppeteer listing scrape returned too few products for catalog discovery', [
                'listing_urls' => $urls->all(),
                'product_count' => count($products),
                'minimum_required' => self::PUPPETEER_MIN_PRODUCTS,
                'error' => $this->lastListingScrapeError,
            ]);

            return null;
        }

        usort(
            $products,
            fn (ScrapedListingProduct $left, ScrapedListingProduct $right): int => $this->compareScrapedProducts($left, $right),
        );

        $mapped = array_map(
            fn (ScrapedListingProduct $product): array => $this->mapScrapedProduct($product),
            $products,
        );

        Log::info('Catalog product discovery used Puppeteer listing scrape', [
            'listing_urls' => $urls->all(),
            'scraped_count' => count($products),
            'selected_count' => count($mapped),
            'product_urls' => array_column($mapped, 'url'),
        ]);

        return $mapped;
    }

    /**
     * @return array<string, mixed>
     */
    private function mapScrapedProduct(ScrapedListingProduct $product): array
    {
        return [
            'name' => $product->name,
            'url' => $product->url,
            'popularity_note' => $this->popularityNoteForScrapedProduct($product),
            'price' => $product->price,
            'list_price' => $product->listPrice,
            'rating' => $product->rating,
            'review_count' => $product->reviewCount,
            'image_url' => $product->imageUrl,
            'discovery_source' => 'puppeteer',
        ];
    }

    private function popularityNoteForScrapedProduct(ScrapedListingProduct $product): string
    {
        $parts = [];

        if ($product->rating !== null) {
            $parts[] = "{$product->rating} stars";
        }

        if ($product->reviewCount !== null) {
            $parts[] = "{$product->reviewCount} reviews";
        }

        if ($product->badges !== []) {
            $parts[] = implode(', ', $product->badges);
        }

        return $parts !== [] ? implode(', ', $parts) : 'Listed on category page';
    }

    private function compareScrapedProducts(ScrapedListingProduct $left, ScrapedListingProduct $right): int
    {
        $reviewComparison = (int) ($right->reviewCount ?? 0) <=> (int) ($left->reviewCount ?? 0);

        if ($reviewComparison !== 0) {
            return $reviewComparison;
        }

        return (float) ($right->rating ?? 0) <=> (float) ($left->rating ?? 0);
    }
}
