<?php

namespace App\Services;

use App\Support\PublicHttp;
use Illuminate\Support\Facades\Log;
use Illuminate\Support\Facades\Process;

/**
 * The HTML a page has once its own JavaScript has run.
 *
 * A plain fetch of lucos.com returns 200 and 1362 bytes: a doctype, a CSP meta
 * tag, a favicon, a viewport and one script tag. Its own CSP gives the reason
 * away, connect-src https://api.lucos.com - the content arrives from an API
 * after the page loads. So the analyser saw a title and nothing else, and every
 * ad written from it read "Visit lucos.com. Explore lucos.com to learn more."
 * Four of Kaushal's five conversations went that way.
 *
 * Only ever a fallback. Rendering costs a browser launch and several seconds,
 * and the three sites where copy suggestions were good - iblockads.org,
 * attestpath.ai, rebates.com - are 13 to 43 times larger and need none of it.
 * LandingPageAnalyzer decides; this only does the work.
 */
class RenderedPageFetcher
{
    /**
     * Rendered HTML, or null when it could not be had.
     *
     * Null rather than an exception: the caller already has the unrendered
     * HTML and a thin page is better than no page. Every failure here is a
     * reason to keep what we have, not to lose the turn.
     */
    public function fetch(string $url): ?string
    {
        if (! config('scraping.puppeteer.enabled')) {
            return null;
        }

        // The same gate the listing scraper uses. A browser fetches far more
        // than an HTTP client does - it follows every subresource the page
        // names - so handing it a private address is worse here than anywhere
        // else in the application.
        if (! PublicHttp::isPublicUrl($url)) {
            return null;
        }

        $script = base_path((string) config('scraping.render_script', 'scripts/render-page.js'));

        if (! is_file($script)) {
            Log::warning('Rendered page fetch skipped: script missing', ['script' => $script]);

            return null;
        }

        $timeout = (int) config('scraping.render_timeout_seconds', 30);

        $result = Process::timeout($timeout + 5)
            ->path(base_path())
            ->run([
                (string) config('scraping.puppeteer.node_binary', 'node'),
                $script,
                $url,
                (string) ($timeout * 1000),
            ]);

        if ($result->failed()) {
            Log::warning('Rendered page fetch failed', [
                'url' => $url,
                'error' => mb_substr(trim($result->errorOutput() ?: $result->output()), 0, 300),
            ]);

            return null;
        }

        $payload = json_decode(trim($result->output()), true);

        if (! is_array($payload) || filled($payload['error'] ?? null)) {
            Log::warning('Rendered page fetch returned no html', [
                'url' => $url,
                'error' => mb_substr((string) ($payload['error'] ?? 'unparseable output'), 0, 300),
            ]);

            return null;
        }

        $html = (string) ($payload['html'] ?? '');

        return $html === '' ? null : $html;
    }
}
