<?php

namespace App\Services\PromptLandingPageEditor;

use App\Models\PromptLandingPageAttachment;
use App\Services\PromptLandingPageHtmlSanitizer;

class SanitizeDocument
{
    public function __construct(private PromptLandingPageHtmlSanitizer $attachmentSanitizer) {}

    /**
     * @param  list<PromptLandingPageAttachment>  $attachments
     * @return array{html: string, css: string, js: string}
     */
    public function sanitize(string $html, string $css, string $js, array $attachments = []): array
    {
        [$html, $cssFromHtml] = $this->extractStyleTags($html);
        $html = $this->sanitizeHtml($html);
        $css = $this->sanitizeCss(trim($css."\n".$cssFromHtml));
        $js = $this->sanitizeJs($js);

        return [
            'html' => $this->attachmentSanitizer->sanitize($html, $attachments),
            'css' => $this->attachmentSanitizer->sanitize($css, $attachments),
            'js' => $this->attachmentSanitizer->sanitize($js, $attachments),
        ];
    }

    /**
     * @return array{0: string, 1: string}
     */
    private function extractStyleTags(string $html): array
    {
        $extraCss = '';
        $rewritten = preg_replace_callback(
            '#<style\b[^>]*>.*?</style>#is',
            function (array $matches) use (&$extraCss): string {
                if ($this->isEditorChromeCss($matches[0])) {
                    return '';
                }

                if (preg_match('#<style\b[^>]*>(.*?)</style>#is', $matches[0], $inner) === 1) {
                    $extraCss .= trim($inner[1])."\n";
                }

                return '';
            },
            $html,
        );

        return [is_string($rewritten) ? $rewritten : $html, $extraCss];
    }

    private function sanitizeHtml(string $html): string
    {
        $html = $this->stripScripts($html);
        $html = (string) preg_replace('#<base\b[^>]*>#i', '', $html);
        $html = (string) preg_replace('#<meta\b[^>]*http-equiv=(["\']?)refresh\1[^>]*>#i', '', $html);
        $html = $this->stripEventHandlers($html);

        // srcdoc is a whole document in an attribute, so nothing below reaches
        // inside it: scripts, handlers and javascript: URLs all survive there.
        $html = (string) preg_replace('/\ssrcdoc\s*=\s*("[^"]*"|\'[^\']*\'|[^\s>]*)/i', '', $html);

        $html = (string) preg_replace('/\sdata-gjs-[a-z0-9-]+="[^"]*"/i', '', $html);
        $html = (string) preg_replace("/\sdata-gjs-[a-z0-9-]+='[^']*'/i", '', $html);
        $html = (string) preg_replace('/\sdraggable="true"/i', '', $html);
        $html = $this->stripGrapesCssRuleNodes($html);

        return $this->neutraliseUrlAttributes($html);
    }

    /**
     * Remove every script, whether or not it is closed the way a regex expects.
     *
     * This required a literal `</script>`. A browser does not: it ends a script
     * at `</script `, `</script/`, `</script\n>` or any `</script` followed by
     * attribute-looking junk, and if the tag is never closed at all it treats the
     * rest of the document as script. So `<script>alert(1)</script >` passed
     * through untouched and ran, and so did a `<script>` left open at the end.
     *
     * Two passes, in this order. Closed pairs first, with the same tolerance a
     * parser has for the closing tag. Then anything still holding an opening
     * `<script` is unterminated by definition, and everything after it would be
     * script to a browser, so it goes with it - including a `<script` that was
     * truncated before its own `>`.
     */
    private function stripScripts(string $html): string
    {
        $html = (string) preg_replace('#<script\b[^>]*>.*?</script\b[^>]*>#is', '', $html);
        $html = (string) preg_replace('#<script\b[^>]*>.*#is', '', $html);

        return (string) preg_replace('#<script\b.*#is', '', $html);
    }

    /**
     * Remove inline event handlers.
     *
     * Nothing removed these, so `<img src=x onerror=alert(1)>` was preserved
     * exactly as written - a bigger hole than the script tags, because it needs
     * no `<script>` at all and survives every other rule here. Inline JavaScript
     * is not part of this document's contract in any case: the editor keeps
     * script in its own field, which is why `<script>` is stripped outright
     * rather than filtered.
     *
     * All three attribute forms, because a value-less or unquoted handler is
     * still a handler: `onerror=alert(1)`, `onerror='alert(1)'`, `onerror`.
     * `data-onclick` is untouched, because the match needs whitespace directly
     * before `on`.
     */
    private function stripEventHandlers(string $html): string
    {
        return (string) preg_replace(
            '/\son[a-z]+(?:\s*=\s*(?:"[^"]*"|\'[^\']*\'|[^\s>]*))?/i',
            '',
            $html,
        );
    }

    /**
     * Point script-bearing URLs at nothing.
     *
     * The quoting was assumed: the pattern required `href="..."`, so
     * `href=javascript:alert(1)` with no quotes was never examined. `data` is
     * here for `<object>`, and `ping` because it is a URL the browser fetches.
     */
    private function neutraliseUrlAttributes(string $html): string
    {
        $rewritten = preg_replace_callback(
            '/\b(href|src|action|formaction|xlink:href|poster|data|ping|background)\s*=\s*'
                .'(?:(["\'])(.*?)\2|([^\s>"\']+))/is',
            function (array $matches): string {
                $quote = $matches[2] ?? '';
                $url = trim($quote === '' ? ($matches[4] ?? '') : $matches[3]);

                if (preg_match('#^(javascript:|vbscript:|data:\s*text/html)#i', $this->undecoded($url)) === 1) {
                    return $matches[1].'="#"';
                }

                return $matches[0];
            },
            $html,
        );

        return is_string($rewritten) ? $rewritten : $html;
    }

    /**
     * An attribute value with its entities and stray whitespace resolved.
     *
     * `java&#115;cript:` and `java\tscript:` are both `javascript:` to a browser
     * and neither matched the check above.
     */
    private function undecoded(string $url): string
    {
        return preg_replace(
            '/[\s\x00-\x1f]+/',
            '',
            html_entity_decode($url, ENT_QUOTES | ENT_HTML5, 'UTF-8'),
        ) ?? $url;
    }

    private function stripGrapesCssRuleNodes(string $html): string
    {
        $previous = null;

        while ($previous !== $html) {
            $previous = $html;
            $next = preg_replace('/<div[^>]*\bid=["\']gjs-css-rules[^"\']*["\'][^>]*>\s*<\/div>/i', '', $html);
            $html = is_string($next) ? $next : $html;
        }

        return $html;
    }

    private function sanitizeCss(string $css): string
    {
        $css = str_replace(['</style', '</STYLE'], ['<\\/style', '<\\/style'], $css);
        $css = (string) preg_replace('/expression\s*\(/i', 'invalid(', $css);
        $css = (string) preg_replace('/javascript\s*:/i', '', $css);
        $css = (string) preg_replace('/-moz-binding/i', 'invalid-binding', $css);
        $css = (string) preg_replace('/(?<![\w-])behavior\s*:/i', 'invalid:', $css);

        return $this->stripEditorChromeCss($css);
    }

    private function isEditorChromeCss(string $css): bool
    {
        return preg_match('/data-gjs-type|\.gjs-|::-webkit-scrollbar/i', $css) === 1;
    }

    private function stripEditorChromeCss(string $css): string
    {
        $css = (string) preg_replace('/\[data-gjs-type[^{]*\{[^{}]*\}/i', '', $css);
        $css = (string) preg_replace('/\.gjs-[A-Za-z0-9_-]*[^{]*\{[^{}]*\}/i', '', $css);

        return trim($css);
    }

    private function sanitizeJs(string $js): string
    {
        return str_replace(['</script', '</SCRIPT'], ['<\\/script', '<\\/script'], $js);
    }
}
