<?php
require_once("lib/php/include/dbPDO.php");

/**
 * Organic search listings for eurekster search pages.
 *
 * Resolution order is Redis, then whale, then Zenserp. The first source that
 * answers is cached, so a repeated term costs no database round trip.
 *
 * The cache is held in two parts. A term maps to the id of the row that
 * answered it, and the listings are held once under that id. Because many
 * terms resolve to the same row, the stored listings are bounded by the number
 * of rows rather than by the number of terms ever searched.
 *
 * A match also records the keyword the matched row belongs to, pointing it at
 * the same id, so searching that keyword later costs no database round trip.
 *
 * Only the listing fields the page renders are cached. The full response is
 * kept in the database, so nothing is lost by holding less in front of it.
 *
 * Only uncached lookups are rate-limited. A throttled client still receives
 * anything already cached, so throttling never empties a page for a term that
 * is already known.
 *
 * Redis is best-effort throughout: when a node is unreachable the lookup runs
 * live rather than raising an error.
 *
 * Targets PHP 7.0+, matching p.eurekster.com (php74).
 */
class EureksterOrganicListings
{
    const TERM_PREFIX       = 'eur:org:term:';
    const DOC_PREFIX        = 'eur:org:doc:';
    const RATE_PREFIX       = 'eur:org:rl:';
    const NO_LISTINGS       = '-';      // stored against a term that resolved to nothing,
                                        // never numeric so it cannot collide with a row id
    const TTL_WITH_LISTINGS = 3888000;  // 45 days
    const TTL_EMPTY         = 86400;    // 1 day, so a barren term is retried sooner
    const RATE_PER_SECOND   = 5;        // uncached lookups per client per second
    const CONNECT_TIMEOUT   = 0.2;
    const ZENSERP_TIMEOUT   = 10;
    const SERP_TYPE         = 'Search';
    const TABLE             = 'eurekster_organic_listings';

    /** fields of a listing the search page renders */
    const LISTING_FIELDS    = ['title', 'url', 'description'];

    /** fields of a local pack entry the search page renders */
    const STORE_FIELDS      = ['title', 'stars', 'street', 'phone'];

    const SOURCE_CACHE     = 'cache';
    const SOURCE_DB        = 'db';
    const SOURCE_API       = 'api';
    const SOURCE_NONE      = 'none';
    const SOURCE_THROTTLED = 'throttled';

    /** @var array host, port that reads are served from */
    private $readNode;

    /** @var array list of host, port pairs every write goes to */
    private $writeNodes;

    /** @var array open redis connections keyed by host:port */
    private $connections = [];

    /** @var PDO|false|null resolved on first use */
    private $db = null;

    /** @var string which source answered the last fetch, one of the SOURCE_ constants */
    private $lastSource = self::SOURCE_NONE;

    /**
     * @param array $config optional overrides: read, write
     */
    public function __construct(array $config = [])
    {
        $this->readNode   = $config['read']  ?? ['127.0.0.1', 22142];
        $this->writeNodes = $config['write'] ?? [['172.16.18.30', 6388], ['172.16.18.31', 6388]];
    }

    /**
     * @param string $term
     * @param string $apiKey      zenserp key
     * @param bool   $useApi      false disables the zenserp fallback
     * @param bool   $bypassCache skip the read but still write, so the previous
     *                            value keeps serving until the fresh one lands
     * @return string JSON carrying the listing fields the page renders, an
     *                empty JSON array when nothing resolved
     */
    public function fetch($term, $apiKey, $useApi = true, $bypassCache = false)
    {
        $term = trim(strtolower($term));

        if ($term === '') {
            return json_encode([]);
        }

        if (!$bypassCache) {
            $cached = $this->cacheGet($term);
            if ($cached !== false) {
                $this->lastSource = self::SOURCE_CACHE;
                return $cached;
            }
        }

        if ($this->rateLimitExceeded()) {
            $this->lastSource = self::SOURCE_THROTTLED;
            return json_encode([]);
        }

        list($id, $matchedKeyword, $blob) = $this->findInDB($term);
        $this->lastSource = ($blob !== '') ? self::SOURCE_DB : self::SOURCE_NONE;

        if ($blob === '' && $useApi) {
            list($id, $matchedKeyword, $blob) = $this->fetchFromZenserp($term, $apiKey);
            $this->lastSource = ($blob !== '') ? self::SOURCE_API : self::SOURCE_NONE;
        }

        $blob = ($blob === '') ? json_encode([]) : $this->slim($blob);

        $this->cachePut($term, $id, $matchedKeyword, $blob);

        return $blob;
    }

    /**
     * Which source answered the last fetch. Intended for debug output.
     *
     * @return string one of the SOURCE_ constants
     */
    public function lastSource()
    {
        return $this->lastSource;
    }

    /**
     * The load balancer appends the real peer to X-Forwarded-For, so the last
     * entry is the only one a client cannot forge. Reading the first entry
     * would let a caller mint a fresh rate limit budget per request.
     *
     * @return string client IP, empty when it cannot be determined
     */
    private function clientIp()
    {
        if (!empty($_SERVER['HTTP_X_FORWARDED_FOR'])) {
            $forwarded = explode(',', $_SERVER['HTTP_X_FORWARDED_FOR']);
            return trim(end($forwarded));
        }

        return $_SERVER['REMOTE_ADDR'] ?? '';
    }

    /**
     * Reduce a response to the listing fields the search page renders.
     *
     * A response carries paid results, related searches, pagination and other
     * sections that are never displayed, and each listing carries fields that
     * are never read. Holding only what is rendered roughly halves what the
     * cache costs per term.
     *
     * The full response stays in the database, so a field that turns out to be
     * needed later is recovered from there rather than lost.
     *
     * The original is returned untouched when it cannot be parsed or when
     * reducing it would leave nothing, so a term that has listings is never
     * mistaken for one that has none.
     *
     * @param  string $blob raw response
     * @return string reduced JSON
     */
    private function slim($blob)
    {
        $decoded = json_decode($blob, true);

        if (empty($decoded['organic']) || !is_array($decoded['organic'])) {
            return $blob;
        }

        $listings = [];

        foreach ($decoded['organic'] as $item) {
            if (!is_array($item)) {
                continue;
            }

            $kept = [];

            foreach (self::LISTING_FIELDS as $field) {
                if (isset($item[$field])) {
                    $kept[$field] = $item[$field];
                }
            }

            if (!empty($item['localPack']) && is_array($item['localPack'])) {
                $stores = [];

                foreach ($item['localPack'] as $store) {
                    if (!is_array($store)) {
                        continue;
                    }

                    $entry = [];
                    foreach (self::STORE_FIELDS as $field) {
                        if (isset($store[$field])) {
                            $entry[$field] = $store[$field];
                        }
                    }

                    if ($entry) {
                        $stores[] = $entry;
                    }
                }

                if ($stores) {
                    $kept['localPack'] = $stores;
                }
            }

            if ($kept) {
                $listings[] = $kept;
            }
        }

        if (!$listings) {
            return $blob;
        }

        $reduced = json_encode(['organic' => $listings]);

        return ($reduced === false) ? $blob : $reduced;
    }

    /**
     * @return string key holding the document id a term resolves to
     */
    private function termKey($term)
    {
        return self::TERM_PREFIX . md5(trim(strtolower($term)));
    }

    /**
     * @return string key holding the listings for a document id
     */
    private function docKey($id)
    {
        return self::DOC_PREFIX . $id;
    }

    /**
     * @param  array $node host, port
     * @return Redis|null null when the node cannot be reached
     */
    private function redis(array $node)
    {
        $id = $node[0] . ':' . $node[1];

        if (array_key_exists($id, $this->connections)) {
            return $this->connections[$id];
        }
        $this->connections[$id] = null;

        if (!class_exists('Redis')) {
            return null;
        }

        try {
            $connection = new Redis();
            if ($connection->pconnect($node[0], (int) $node[1], self::CONNECT_TIMEOUT)) {
                $this->connections[$id] = $connection;
            }
        } catch (Exception $e) {
            $this->connections[$id] = null;
        }

        return $this->connections[$id];
    }

    /**
     * Resolve a term through the id of the row that answered it.
     *
     * Listings are held once per row id, so terms that resolve to the same row
     * share a single copy instead of each holding their own.
     *
     * @return string|false the stored blob, false on a miss or when redis is down
     */
    private function cacheGet($term)
    {
        $redis = $this->redis($this->readNode);
        if (!$redis) {
            return false;
        }

        try {
            $id = $redis->get($this->termKey($term));

            if ($id === false) {
                return false;
            }

            if ($id === self::NO_LISTINGS) {
                return json_encode([]);
            }

            // a document dropped from under its term reads as a miss, so the
            // lookup runs again rather than the page losing its listings
            $blob = $redis->get($this->docKey($id));

            return ($blob === '') ? false : $blob;
        } catch (Exception $e) {
            return false;
        }
    }

    /**
     * Write to every node so both front ends serve the same content.
     *
     * The listings are stored against the row id and the term stores only that
     * id, so a repeat term that resolves to the same row costs one short
     * pointer rather than another copy of the listings.
     *
     * The keyword the matched row belongs to is pointed at the same id, so a
     * later search for that keyword is answered from the cache rather than
     * repeating the lookup that has already been done here.
     *
     * @param string $term           the searched term, already normalised
     * @param string $id             row id the term resolved to
     * @param string $matchedKeyword keyword of that row, empty when not known
     * @param string $blob           listings to store
     */
    private function cachePut($term, $id, $matchedKeyword, $blob)
    {
        $decoded     = json_decode($blob, true);
        $hasListings = !empty($decoded['organic']);

        $termKey  = $this->termKey($term);
        $matched  = trim(strtolower($matchedKeyword));
        $aliasKey = ($hasListings && $matched !== '' && $matched !== $term)
            ? $this->termKey($matched)
            : '';

        foreach ($this->writeNodes as $node) {
            $redis = $this->redis($node);
            if (!$redis) {
                continue;
            }
            try {
                if ($hasListings) {
                    $redis->setex($this->docKey($id), self::TTL_WITH_LISTINGS, $blob);
                    $redis->setex($termKey, self::TTL_WITH_LISTINGS, $id);
                    if ($aliasKey !== '') {
                        $redis->setex($aliasKey, self::TTL_WITH_LISTINGS, $id);
                    }
                } else {
                    $redis->setex($termKey, self::TTL_EMPTY, self::NO_LISTINGS);
                }
            } catch (Exception $e) {
                // one node failing must not cost the visitor their results
            }
        }
    }

    /**
     * Count this lookup against the client's budget for the current second.
     *
     * The counter is incremented on every node, which keeps them in step and
     * means whichever node is read reports the true count.
     *
     * Fails open, since losing the counter must not take the feature down.
     *
     * @return bool true once the budget is spent
     */
    private function rateLimitExceeded()
    {
        $ip = $this->clientIp();
        if ($ip === '') {
            return false;
        }

        $key  = self::RATE_PREFIX . $ip . ':' . time();
        $used = 0;

        foreach ($this->writeNodes as $node) {
            $redis = $this->redis($node);
            if (!$redis) {
                continue;
            }
            try {
                $count = $redis->incr($key);
                if ($count === 1) {
                    $redis->expire($key, 2);
                }
                // a node that was briefly down lags behind, so trust the highest
                $used = max($used, $count);
            } catch (Exception $e) {
                continue;
            }
        }

        return $used > self::RATE_PER_SECOND;
    }

    /**
     * @return PDO|false
     */
    private function db()
    {
        if ($this->db === null) {
            $this->db = CommonPDOConnector::connect('ConnectToWhale');
        }

        return $this->db;
    }

    /**
     * @return array row id, the keyword that row belongs to, and the stored blob,
     *               all empty when there is no usable row
     */
    private function findInDB($term)
    {
        $db = $this->db();
        if (!$db) {
            return ['', '', ''];
        }

        $sql = "SELECT id, keyword, result FROM " . self::TABLE . " WHERE MATCH(keyword) AGAINST (? IN NATURAL LANGUAGE MODE) AND valid = 1 AND serp_type = ? LIMIT 1";

        $statement = $db->prepare($sql);
        $statement->execute([$term, self::SERP_TYPE]);
        $row = $statement->fetch(PDO::FETCH_ASSOC);

        if (!$row || empty($row['result'])) {
            return ['', '', ''];
        }

        $stored  = stripslashes($row['result']);
        $decoded = json_decode($stored, true);

        if (empty($decoded['organic'])) {
            return ['', '', ''];
        }

        return [(string) $row['id'], (string) $row['keyword'], $stored];
    }

    /**
     * The row is created from the searched term itself, so there is no second
     * keyword to point at it.
     *
     * @return array row id, an empty keyword, and the api response, all empty
     *               when the response carried no listings
     */
    private function fetchFromZenserp($term, $apiKey)
    {
        $url = 'https://app.zenserp.com/api/v2/search?apikey=' . $apiKey
             . '&q=' . urlencode($term)
             . '&device=desktop&location=Manhattan,New%20York,United%20States&num=100';

        $response = $this->httpGet($url, self::ZENSERP_TIMEOUT);
        $decoded  = json_decode($response, true);

        if (empty($decoded['organic'])) {
            return ['', '', ''];
        }

        $id = $this->storeInDB($term, $response);

        // without a row id the listings cannot be shared, but they are still
        // worth holding under a key of their own rather than paying for the
        // same api call again
        if ($id === '') {
            $id = 't' . md5($term);
        }

        return [$id, '', $response];
    }

    /**
     * @return string id of the inserted row, empty when the row was not stored
     *                or its id could not be determined
     */
    private function storeInDB($term, $response)
    {
        $db = $this->db();
        if (!$db) {
            return '';
        }

        try {
            // the result column is utf8mb3, so 4 byte characters break the insert
            $storable = preg_replace('/[\x{10000}-\x{10FFFF}]/u', '', $response);

            $sql = "INSERT INTO " . self::TABLE . " (keyword, result, serp_type) VALUES (?, ?, ?)";
            $statement = $db->prepare($sql);
            $statement->execute([$term, addslashes($storable), self::SERP_TYPE]);

            // a driver that cannot report the id gives 0, which would make every
            // such row share one document key
            $id = (string) $db->lastInsertId();

            return ($id === '' || $id === '0') ? '' : $id;
        } catch (Exception $e) {
            // a failed insert must not cost the visitor their results
            return '';
        }
    }

    /**
     * @return string response body, empty string on failure
     */
    private function httpGet($url, $timeout)
    {
        $curl = curl_init($url);
        curl_setopt($curl, CURLOPT_RETURNTRANSFER, true);
        curl_setopt($curl, CURLOPT_SSL_VERIFYPEER, false);
        curl_setopt($curl, CURLOPT_TIMEOUT, $timeout);

        $response = curl_exec($curl);
        curl_close($curl);

        return is_string($response) ? $response : '';
    }
}
