<?php

namespace Tests\Feature;

use App\DTOs\LandingPageAnalysis;
use App\Models\AllowedAdAccount;
use App\Models\Avatar;
use App\Models\Voice;
use App\Services\AI\CatalogWebsiteResearch;
use App\Services\AI\GeminiVeoVertexVideoProvider;
use App\Services\AI\LaravelAiCreativeBriefResolver;
use App\Services\LandingPageAnalyzer;
use App\Services\Platforms\Support\Guardrails;
use App\Support\VideoDialogue;
use Illuminate\Foundation\Testing\RefreshDatabase;
use Illuminate\Support\Facades\Http;
use Illuminate\Support\Facades\Storage;
use InvalidArgumentException;
use PHPUnit\Framework\Attributes\DataProvider;
use Tests\TestCase;

/**
 * Eight findings from the repo review, pinned.
 *
 * Three of them turned out not to be live defects when checked against the
 * running code, and those are recorded as what they actually are: a guard that
 * cannot regress, a limit with no margin left, and an unreachable fallback that
 * would fail silently if it ever became reachable. Saying so is the point -
 * an item marked fixed that was never broken is worth as little as one the other
 * way round.
 */
class EightMoreFromTheReviewTest extends TestCase
{
    use RefreshDatabase;

    // -------------------------------------------- 22: page code as keywords

    /**
     * Scripts and stylesheets are not what the page is about.
     *
     * string(//body) returns every descendant's text, so an inline analytics
     * snippet or a block of CSS - on almost every landing page - was counted.
     * "function", "datalayer", "gtag", "webkit" and "transition" all became
     * keywords, and GooglePublisher sends these to Google as PHRASE match keywords
     * on a live Search campaign. The same text is what ad copy is written from, so
     * the model was being shown JavaScript and asked what the page sells.
     */
    public function test_page_code_is_not_read_as_page_content(): void
    {
        $analysis = $this->analyze(
            '<html><head><title>Adblocker for Chrome</title>'
            .'<style>.hero{background:transparent;transition:transform 180ms ease;display:flex}</style></head>'
            .'<body><h1>Block adblocker-breaking ads</h1>'
            .'<p>Our adblocker stops trackers. The adblocker is free.</p>'
            .'<script>window.dataLayer=window.dataLayer||[];function gtag(){dataLayer.push(arguments)}'
            .'gtag("js",new Date());document.addEventListener("DOMContentLoaded",function(){});</script>'
            .'<style>.footer{webkit-appearance:none}</style></body></html>'
        );

        foreach (['datalayer', 'function', 'gtag', 'webkit', 'transition', 'domcontentloaded', 'appearance'] as $code) {
            $this->assertNotContains($code, $analysis->keywords, "[{$code}] came out of the page's code");
            $this->assertStringNotContainsString($code, strtolower($analysis->text), "[{$code}] is in the text ad copy is written from");
        }
    }

    /** And the real content still is. */
    public function test_the_actual_content_still_comes_through(): void
    {
        $analysis = $this->analyze(
            '<html><head><title>Adblocker for Chrome</title></head><body>'
            .'<h1>Block intrusive ads</h1><p>Our adblocker stops trackers. The adblocker is free.</p>'
            .'<script>var x=1;</script></body></html>'
        );

        $this->assertContains('adblocker', $analysis->keywords);
        $this->assertStringContainsString('stops trackers', $analysis->text);
    }

    /** A noscript fallback is furniture too, not content. */
    public function test_a_noscript_block_is_not_content(): void
    {
        $analysis = $this->analyze(
            '<html><head><title>Offers</title></head><body><h1>Winter offers</h1>'
            .'<noscript>Please enable javascript to continue browsing this website.</noscript>'
            .'<p>Winter offers on every winter coat.</p></body></html>'
        );

        $this->assertStringNotContainsString('enable javascript', strtolower($analysis->text));
    }

    private function analyze(string $html): LandingPageAnalysis
    {
        config(['scraping.puppeteer.enabled' => false]);
        Http::fake(['*' => Http::response($html)]);

        return app(LandingPageAnalyzer::class)->analyze('https://example.com/landing');
    }

    // ------------------------------- 23: identifiers MySQL will not accept

    /**
     * No migration generates an identifier MySQL would refuse.
     *
     * Not a defect today: every over-long default in this repo already carries an
     * explicit short name, and the 325 identifiers in the live MariaDB schema are
     * all inside the limit. This exists because nothing would catch the next one -
     * the suite runs on sqlite, which has no 64-character limit, so an over-long
     * name passes every test and fails on deploy.
     *
     * Four names sit at exactly 64, so there is no margin: one longer column name
     * on any of those tables breaks the migration.
     */
    public function test_no_migration_names_an_index_mysql_would_refuse(): void
    {
        $tooLong = [];

        foreach (glob(base_path('database/migrations/*.php')) as $file) {
            foreach ($this->generatedIdentifiers((string) file_get_contents($file)) as $identifier) {
                if (strlen($identifier) > 64) {
                    $tooLong[] = basename($file).': '.$identifier.' ('.strlen($identifier).')';
                }
            }
        }

        $this->assertSame([], $tooLong, 'MySQL refuses an identifier over 64 characters.');
    }

    /**
     * Index names a migration leaves to Laravel to generate.
     *
     * Only the unnamed ones: an explicit name is the fix, so flagging a call that
     * already has one is how a check like this gets ignored. My first pass at this
     * did exactly that and reported 16 false positives.
     *
     * @return list<string>
     */
    private function generatedIdentifiers(string $source): array
    {
        $identifiers = [];

        foreach (preg_split('/Schema::(?:create|table)\(/', $source) as $chunk) {
            if (preg_match("/^\s*'([a-z0-9_]+)'/", $chunk, $table) !== 1) {
                continue;
            }

            // Multi-column index and unique calls, with everything up to the
            // closing bracket so a second argument is visible.
            foreach (['unique', 'index'] as $kind) {
                preg_match_all('/->'.$kind.'\(\s*\[(.*?)\]([^;]*?)\)/s', $chunk, $calls, PREG_SET_ORDER);

                foreach ($calls as $call) {
                    if (str_contains($call[2], "'")) {
                        continue; // an explicit name was given
                    }

                    $columns = implode('_', preg_split('/\W+/', trim($call[1]), flags: PREG_SPLIT_NO_EMPTY) ?: []);
                    $identifiers[] = $table[1].'_'.$columns.'_'.$kind;
                }
            }

            // foreignId()->constrained() without an indexName.
            preg_match_all('/->foreignId\(\s*\'([a-z0-9_]+)\'\s*\)(.*?);/s', $chunk, $keys, PREG_SET_ORDER);

            foreach ($keys as $key) {
                if (str_contains($key[2], 'indexName')) {
                    continue;
                }

                $identifiers[] = $table[1].'_'.$key[1].'_foreign';
            }
        }

        return $identifiers;
    }

    // ------------------------------------ 24: a video of the wrong length

    /**
     * A duration with no stage plan is refused rather than rendered wrong.
     *
     * The extend provider builds 8 seconds plus 7 per extend, and defaulted to one
     * extend for any duration it had no entry for - so an 8-second request would
     * have produced 15 seconds of video, stored and published as the 8 that was
     * asked for, with nothing downstream checking the length.
     *
     * Unreachable through VideoAvatarProviderFactory, which only routes a duration
     * here when the map has it. One line in that class away from being wrong, and
     * silent when it is.
     */
    #[DataProvider('unplannedDurations')]
    public function test_a_duration_with_no_stage_plan_is_refused(int $duration): void
    {
        config([
            'creative.vertex_extend_durations' => [15 => 1, 30 => 3],
            'services.vertex_ai.project_id' => 'test-project',
            'services.vertex_ai.credentials_path' => base_path('composer.json'),
        ]);

        $this->expectException(InvalidArgumentException::class);
        $this->expectExceptionMessage('No extend plan');

        app(GeminiVeoVertexVideoProvider::class)->render(
            new Avatar, new Voice, [], [], $duration, '16:9',
        );
    }

    /** @return list<array{0: int}> */
    public static function unplannedDurations(): array
    {
        return ['eight' => [8], 'twenty' => [20], 'sixty' => [60]];
    }

    /** The word budget and the stages agree for the durations that are planned. */
    public function test_the_planned_durations_add_up(): void
    {
        config(['creative.vertex_extend_durations' => [15 => 1, 30 => 3]]);

        foreach ([15 => [8, 7], 30 => [8, 7, 7, 7]] as $duration => $stages) {
            $perStage = array_sum(array_map(
                fn (int $stage): int => VideoDialogue::wordBudget($stage),
                $stages,
            ));

            $this->assertSame(
                VideoDialogue::wordBudget($duration),
                $perStage,
                "the {$duration}s budget does not match what its stages can carry",
            );
        }
    }

    // ---------------------------- 25: a file written into a directory

    /**
     * A non-image response is not stored as one.
     *
     * The Content-Type was read as `header('Content-Type', 'image/jpeg')`, which
     * looks like a default but is not - Response::header() takes one argument - so
     * a non-image response was never considered. Then Str::after('text/html',
     * 'image/') returns the whole string when the needle is absent, so the
     * extension became `text/html` and the path became
     * `creatives/brief-images/<uuid>.text/html`: a directory named `<uuid>.text`
     * holding a file called `html`, with an HTML error page inside it, recorded on
     * the brief as a reference image.
     */
    #[DataProvider('nonImageTypes')]
    public function test_a_non_image_is_not_stored(string $contentType): void
    {
        Storage::fake('public');
        Http::fake(['*' => Http::response('<html>Not found</html>', 200, ['Content-Type' => $contentType])]);

        $this->assertNull($this->download('https://example.com/not-an-image'));
        $this->assertSame([], Storage::disk('public')->allFiles('creatives/brief-images'));
        $this->assertSame([], Storage::disk('public')->directories('creatives/brief-images'));
    }

    /** @return list<array{0: string}> */
    public static function nonImageTypes(): array
    {
        return [
            'html' => ['text/html'],
            'html with charset' => ['text/html; charset=utf-8'],
            'json' => ['application/json'],
            'nothing at all' => [''],
        ];
    }

    /** A real image is still stored, with a sane extension. */
    #[DataProvider('imageTypes')]
    public function test_an_image_is_stored_with_its_extension(string $contentType, string $extension): void
    {
        Storage::fake('public');
        Http::fake(['*' => Http::response('binary', 200, ['Content-Type' => $contentType])]);

        [$path] = $this->download('https://example.com/hero');

        $this->assertStringEndsWith('.'.$extension, $path);
        $this->assertStringNotContainsString('/', substr($path, strrpos($path, '.')));
        Storage::disk('public')->assertExists($path);
    }

    /** @return list<array{0: string, 1: string}> */
    public static function imageTypes(): array
    {
        return [
            'jpeg' => ['image/jpeg', 'jpg'],
            'png' => ['image/png', 'png'],
            'webp with charset' => ['image/webp; charset=binary', 'webp'],
            'uppercase' => ['IMAGE/PNG', 'png'],
        ];
    }

    private function download(string $url): ?array
    {
        $resolver = app(LaravelAiCreativeBriefResolver::class);
        $method = new \ReflectionMethod($resolver, 'downloadAndStoreImage');

        return $method->invoke($resolver, $url);
    }

    // ------------------------------- 26: a filter that filtered nothing

    /**
     * A listing token matches the word it came from.
     *
     * The token had its trailing s chopped, and the matcher requires a whole-word
     * match, so "shoes" became "shoe" and "shoe" does not match "shoes" - the s
     * after it is not a word boundary. "dress" became "dres", which matches
     * nothing at all. Every token has to match for a link to be kept, and fewer
     * than five matches makes the whole filter return the unfiltered list, so the
     * result was no filter at all on any listing whose words end in s.
     */
    #[DataProvider('tokenAndPage')]
    public function test_a_listing_token_matches_its_own_page(string $token, string $linkText): void
    {
        $this->assertTrue(
            $this->linkMatches($token, $linkText),
            "[{$token}] did not match [{$linkText}]",
        );
    }

    /** @return list<array{0: string, 1: string}> */
    public static function tokenAndPage(): array
    {
        return [
            'plural token, plural page' => ['shoes', 'Running shoes for men'],
            'plural token, singular page' => ['shoes', 'Mens running shoe'],
            'singular token, plural page' => ['shoe', 'Running shoes for men'],
            'double s stays whole' => ['dress', 'Summer dress in linen'],
            'double s against its plural' => ['dress', 'Summer dresses in linen'],
            'plural of a double s' => ['dresses', 'Summer dress in linen'],
            'in the url' => ['jackets', 'Winter range /collections/jacket/quilted'],
            'ordinary word' => ['linen', 'Summer dress in linen'],
        ];
    }

    /** And it still refuses a link that is about something else. */
    public function test_an_unrelated_link_is_still_refused(): void
    {
        $this->assertFalse($this->linkMatches('shoes', 'Leather handbags and purses'));
        $this->assertFalse($this->linkMatches('dress', 'Running shoes for men'));
    }

    /** The token itself is no longer mangled on the way in. */
    public function test_a_token_is_not_chopped(): void
    {
        $research = app(CatalogWebsiteResearch::class);
        $method = new \ReflectionMethod($research, 'listingTokenWords');

        $tokens = $method->invoke($research, 'Summer dress and shoes');

        $this->assertContains('dress', $tokens, 'dress was chopped to dres');
        $this->assertNotContains('dres', $tokens);
    }

    private function linkMatches(string $token, string $linkText): bool
    {
        $research = app(CatalogWebsiteResearch::class);
        $method = new \ReflectionMethod($research, 'linkMatchesAllListingTokens');

        return $method->invoke($research, ['name' => $linkText, 'url' => ''], [$token]);
    }

    // ------------------------- 27, 28, 29: the guardrail one-liners

    /**
     * A wildcard row in the table opens the platform, as it says it does.
     *
     * isAllowed() compared ids only, so a row holding `*` normalised to `*`,
     * matched no real id, and the row meant to open the platform closed it -
     * while Guardrails::allows() honoured the same wildcard from config. So
     * `meta:*` in the env worked and a `*` row in allowed_ad_accounts did the
     * opposite of what it says.
     */
    public function test_a_wildcard_row_allows_every_account_on_that_platform(): void
    {
        AllowedAdAccount::create(['platform' => 'meta', 'account_id' => '*', 'is_active' => true]);

        $this->assertTrue(AllowedAdAccount::isAllowed('act_1950320145707342', 'meta'));
        $this->assertTrue(AllowedAdAccount::isAllowed('123', 'meta'));

        // And says nothing about the others.
        $this->assertFalse(AllowedAdAccount::isAllowed('508765432', 'linkedin'));
    }

    /** A real row still only allows its own account. */
    public function test_a_named_row_allows_only_that_account(): void
    {
        AllowedAdAccount::create(['platform' => 'meta', 'account_id' => 'act_111', 'is_active' => true]);

        $this->assertTrue(AllowedAdAccount::isAllowed('act_111', 'meta'));
        $this->assertTrue(AllowedAdAccount::isAllowed('111', 'meta'));
        $this->assertFalse(AllowedAdAccount::isAllowed('act_222', 'meta'));
    }

    /**
     * act_ belongs to Meta and to nothing else.
     *
     * It was the default arm, so every platform without a case of its own got it:
     * a TikTok advertiser id was stored and compared as act_7123456789, and so
     * would Taboola's. Both sides pass through the same method so the comparison
     * still worked, which is why it survived - but every message listing permitted
     * accounts printed Meta's prefix on another platform's id.
     */
    #[DataProvider('accountIds')]
    public function test_an_account_id_is_normalised_for_its_own_platform(string $id, string $platform, string $expected): void
    {
        $this->assertSame($expected, AllowedAdAccount::normalizeAccountId($id, $platform));
    }

    /** @return list<array{0: string, 1: string, 2: string}> */
    public static function accountIds(): array
    {
        return [
            'meta, prefixed' => ['act_123', 'meta', 'act_123'],
            'meta, bare' => ['123', 'meta', 'act_123'],
            'meta, shouting' => ['ACT_123', 'meta', 'act_123'],
            'google strips dashes' => ['456-789-0123', 'google', '4567890123'],
            'linkedin takes the last segment' => ['urn:li:sponsoredAccount:508', 'linkedin', '508'],
            'tiktok keeps its own id' => ['7123456789', 'tiktok', '7123456789'],
            'taboola keeps its own id' => ['1234567', 'taboola', '1234567'],
            'the wildcard is not an account' => ['*', 'meta', '*'],
            'nor on another platform' => ['*', 'tiktok', '*'],
        ];
    }

    /**
     * A URN in the allow list is recognised whatever case it is written in.
     *
     * Both checks were case-sensitive, so `URN:LI:sponsoredAccount:5087` failed the
     * URN test, fell into the prefixed-entry branch, split at the first colon into
     * a prefix of `urn`, matched no platform, and was dropped from the allow list
     * without a word. The administrator had permitted that account and the guard
     * refused it.
     */
    #[DataProvider('urnSpellings')]
    public function test_a_urn_entry_is_read_whatever_case_it_is_in(string $entry): void
    {
        config(['platforms.guardrails.allowed_ad_accounts' => [$entry]]);

        $this->assertTrue(
            app(Guardrails::class)->allows('508765432', 'linkedin'),
            "[{$entry}] was dropped from the allow list",
        );
    }

    /** @return list<array{0: string}> */
    public static function urnSpellings(): array
    {
        return [
            'lower' => ['urn:li:sponsoredAccount:508765432'],
            'upper' => ['URN:LI:sponsoredAccount:508765432'],
            'mixed' => ['Urn:Li:SponsoredAccount:508765432'],
            'prefixed instead' => ['linkedin:508765432'],
        ];
    }

    /** A Meta prefix is read either way too, and stays Meta's. */
    public function test_a_shouted_meta_entry_is_still_meta(): void
    {
        config(['platforms.guardrails.allowed_ad_accounts' => ['ACT_1950320145707342']]);

        $guardrails = app(Guardrails::class);

        $this->assertTrue($guardrails->allows('act_1950320145707342', 'meta'));
        $this->assertFalse($guardrails->allows('1950320145707342', 'google'));
    }
}
