#!/usr/bin/env bash
# Seed commit intelligence on dev without a full repo-sync re-index.
#
# Copies commit_analyses from a JSON export (mongoexport --jsonArray), remaps
# repo_id / snapshot_id, upserts into dev Mongo, and publishes commit.analysis.ready
# events so embedding-engine Celery workers build Qdrant vectors.
#
# 1) Export on local Mac:
#
#   docker exec mongo-db mongoexport \
#     --db=adpilot_commit_intel \
#     --collection=commit_analyses \
#     --query='{"repo_id":"ad/data-pipeline.leisure.com"}' \
#     --jsonArray --out=/tmp/commits.json
#
#   docker cp mongo-db:/tmp/commits.json ./scripts/fixtures/data-pipeline-commits.json
#
# 2) On dev (from repo root; uses scripts/fixtures/data-pipeline-commits.json by default):
#
#   ./scripts/seed_commit_intel_dev.sh
#
# Or with explicit path:
#
#   ./scripts/seed_commit_intel_dev.sh \
#     --input ./scripts/fixtures/data-pipeline-commits.json
#
# Requires: docker, jq

set -euo pipefail

SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
DEFAULT_INPUT="${SCRIPT_DIR}/fixtures/data-pipeline-commits.json"
DEFAULT_REPO_ID="bitbucket_server:AD/data-pipeline.leisure.com"
DEFAULT_SNAPSHOT_ID="snap_dc8bb83370e6c152"

INPUT=""
REPO_ID=""
SNAPSHOT_ID=""
MONGO_CONTAINER="mongo-db"
REDIS_CONTAINER="redis-server"
STREAM="commit.analysis.ready"
SKIP_REDIS=0
SKIP_INDEXING_RUN=0
DRY_RUN=0

usage() {
  sed -n '2,24p' "$0" | sed 's/^# \?//'
  cat <<'EOF'

Options:
  --input PATH              JSON array from mongoexport (default: fixtures/data-pipeline-commits.json)
  --repo-id ID              Target dev repo_id (default: bitbucket_server:AD/data-pipeline.leisure.com)
  --snapshot-id ID          Target dev snapshot_id (default: snap_dc8bb83370e6c152)
  --mongo-container NAME    Default: mongo-db
  --redis-container NAME    Default: redis-server
  --stream NAME             Default: commit.analysis.ready
  --skip-redis              Only upsert Mongo
  --skip-indexing-run       Do not patch adpilot_repo_sync.indexing_runs
  --dry-run                 Print actions without writing
  -h, --help                Show this help
EOF
}

while [[ $# -gt 0 ]]; do
  case "$1" in
    --input) INPUT="$2"; shift 2 ;;
    --repo-id) REPO_ID="$2"; shift 2 ;;
    --snapshot-id) SNAPSHOT_ID="$2"; shift 2 ;;
    --mongo-container) MONGO_CONTAINER="$2"; shift 2 ;;
    --redis-container) REDIS_CONTAINER="$2"; shift 2 ;;
    --stream) STREAM="$2"; shift 2 ;;
    --skip-redis) SKIP_REDIS=1; shift ;;
    --skip-indexing-run) SKIP_INDEXING_RUN=1; shift ;;
    --dry-run) DRY_RUN=1; shift ;;
    -h|--help) usage; exit 0 ;;
    *) echo "Unknown option: $1" >&2; usage >&2; exit 1 ;;
  esac
done

INPUT="${INPUT:-$DEFAULT_INPUT}"
REPO_ID="${REPO_ID:-$DEFAULT_REPO_ID}"
SNAPSHOT_ID="${SNAPSHOT_ID:-$DEFAULT_SNAPSHOT_ID}"

echo "Using input: $INPUT"
echo "Target repo: $REPO_ID"
echo "Target snapshot: $SNAPSHOT_ID"

if [[ ! -f "$INPUT" ]]; then
  echo "Error: input file not found: $INPUT" >&2
  echo "Hint: pull latest repo (fixtures at scripts/fixtures/data-pipeline-commits.json) or pass --input PATH" >&2
  exit 1
fi

if ! command -v jq >/dev/null 2>&1; then
  echo "Error: jq is required." >&2
  exit 1
fi

if ! jq -e 'type == "array"' "$INPUT" >/dev/null 2>&1; then
  echo "Error: input must be a JSON array." >&2
  exit 1
fi

COMMIT_COUNT="$(jq 'length' "$INPUT")"
if [[ "$COMMIT_COUNT" -eq 0 ]]; then
  echo "Error: input array is empty." >&2
  exit 1
fi

WORKDIR="$(mktemp -d)"
trap 'rm -rf "$WORKDIR"' EXIT
REMAPPED="$WORKDIR/remapped.json"

# Remap documents for dev repo/snapshot (deterministic ids per commit_sha).
jq --arg repo "$REPO_ID" --arg snap "$SNAPSHOT_ID" '
  map(
    if (.commit_sha // "" | length) == 0 then
      error("commit_sha missing in source document")
    else
      .
    end
    | del(._id)
    | .repo_id = $repo
    | .snapshot_id = $snap
    | .commit_sha as $sha
    | ($sha[0:8]) as $short
    | .analysis_id = ("analysis_manual_" + $short)
    | .event_id = ("evt_manual_seed_" + $short)
    | .delta_id = (.delta_id // ("delta_manual_" + $short))
    | (if (.summary // "" | gsub("^\\s+|\\s+$"; "") | length) == 0 then
        (if (.commit_message // "" | gsub("^\\s+|\\s+$"; "") | length) > 0 then
          .commit_message
        else
          "Commit " + $short
          + (if (.changed_files // [] | length) > 0 then
              " Files: " + ((.changed_files // [])[0:8] | join(", "))
            else "" end)
        end)
      else .summary end) as $summary
    | .summary = $summary
    | (if (.commit_message // "" | gsub("^\\s+|\\s+$"; "") | length) == 0 then
        ($summary | split("\n")[0][0:500])
      else .commit_message end) as $msg
    | .commit_message = $msg
  )
' "$INPUT" > "$REMAPPED"

echo "Loaded $COMMIT_COUNT commits; seeding to $REPO_ID / $SNAPSHOT_ID"

mongo_upsert() {
  if [[ "$DRY_RUN" -eq 1 ]]; then
    echo "[dry-run] would upsert $COMMIT_COUNT commit_analyses documents"
    return 0
  fi

  local docs_json
  docs_json="$(jq -c '.' "$REMAPPED")"

  docker exec -i "$MONGO_CONTAINER" mongosh adpilot_commit_intel --quiet <<EOF
const docs = ${docs_json};
const ops = docs.map((doc) => ({
  updateOne: {
    filter: { analysis_id: doc.analysis_id },
    update: { \$set: doc },
    upsert: true,
  },
}));
const res = db.commit_analyses.bulkWrite(ops, { ordered: false });
printjson({ matched: res.matchedCount, modified: res.modifiedCount, upserted: res.upsertedCount });
EOF
}

patch_indexing_run() {
  if [[ "$SKIP_INDEXING_RUN" -eq 1 ]]; then
    return 0
  fi
  if [[ "$DRY_RUN" -eq 1 ]]; then
    echo "[dry-run] would patch indexing_run expected_commits=$COMMIT_COUNT"
    return 0
  fi

  docker exec -i "$MONGO_CONTAINER" mongosh --quiet <<EOF
const filter = { repo_id: $(jq -Rn --arg v "$REPO_ID" '$v'), snapshot_id: $(jq -Rn --arg v "$SNAPSHOT_ID" '$v') };
const update = {
  \$set: {
    expected_commits: $COMMIT_COUNT,
    expected_commit_deltas: $COMMIT_COUNT,
    "stages.commit_intel": "completed",
  },
};
const res = db.getSiblingDB("adpilot_repo_sync").indexing_runs.updateOne(filter, update);
printjson({ matched: res.matchedCount, modified: res.modifiedCount });
EOF
}

publish_redis_events() {
  if [[ "$SKIP_REDIS" -eq 1 ]]; then
    echo "Skipped Redis publish (--skip-redis)."
    return 0
  fi

  local i payload entry_id analysis_id
  for ((i = 0; i < COMMIT_COUNT; i++)); do
    payload="$(jq -c --arg repo "$REPO_ID" '.[$i] | {
      event_id,
      event_version: "v1",
      repo_id: $repo,
      commit_sha,
      analysis_id,
      base_commit_sha,
      summary,
      changed_files: (.changed_files // []),
      impacted_symbols: (.impacted_symbols // []),
      graph_schema_version: "v1"
    }' --argjson i "$i" "$REMAPPED")"
    analysis_id="$(jq -r --argjson i "$i" '.[$i].analysis_id' "$REMAPPED")"

    if [[ "$DRY_RUN" -eq 1 ]]; then
      echo "[dry-run] XADD $STREAM analysis_id=$analysis_id"
      continue
    fi

    entry_id="$(docker exec "$REDIS_CONTAINER" redis-cli XADD "$STREAM" '*' payload "$payload")"
    printf 'published %s… -> %s\n' "${analysis_id:0:40}" "$entry_id"
    sleep 0.05
  done

  cat <<EOF

Embedding jobs queued. Watch progress:
  docker logs -f embedding-engine-service 2>&1 | grep -i commit
  docker exec $MONGO_CONTAINER mongosh adpilot_indexing --quiet --eval \\
    'db.embedding_records.countDocuments({ repo_id: /data-pipeline/i, source_type: "commit" })'
EOF
}

mongo_upsert
patch_indexing_run
publish_redis_events

echo "Done."
