#!/usr/bin/env bash
# LLM-Gate's one command. `./gate help` for the list.
#
# It is deliberately thin: everything it does is docker compose plus the two things that are easy to get wrong
# by hand — minting a key and taking one away, each of which also reloads the door.
set -euo pipefail
cd "$(dirname "$0")"

KEYS=keys.conf
log()  { printf '%s  %s\n' "$(date +%H:%M:%S)" "$*"; }
die()  { printf 'gate: %s\n' "$*" >&2; exit 1; }
dc()   { docker compose "$@"; }

envval() { # envval NAME — read one value out of .env without sourcing it
    [ -f .env ] || return 0
    grep -E "^$1=" .env | head -1 | cut -d= -f2- | tr -d '\r' | tr -d "\"'"
}

first_run() {
    [ -f .env ]   || { cp .env.example .env; log "created .env from .env.example"; }
    [ -f "$KEYS" ] || { cp keys.conf.example "$KEYS"
                        sed -i '/REPLACE_ME/d' "$KEYS"
                        log "created $KEYS — mint a key with: ./gate keys add <name>"; }
    mkdir -p "$(envval MODELS_DIR | sed 's|^$|./models|')"
}

port() { local p; p=$(envval GATE_PORT); echo "${p:-11444}"; }

status() {
    dc ps --format 'table {{.Service}}\t{{.Status}}\t{{.Ports}}'
    echo
    printf 'health:   '; curl -s -m 10 "http://127.0.0.1:$(port)/healthz" || echo "(no answer)"
    echo
    echo 'models:'; dc exec -T llm ollama list 2>/dev/null || echo '  (the model container is not up)'
    echo; echo 'callers:'; keys list
}

keys() {
    local action=${1:-list}; shift || true
    case $action in
        list)
            printf '  %-24s %s\n' CALLER KEY
            sed -n 's/^"\([^"]*\)"[[:space:]]*"\([^"]*\)";.*/\2 \1/p' "$KEYS" | while read -r name key; do
                printf '  %-24s %s…%s\n' "$name" "${key%"${key#?????????}"}" "${key#"${key%????}"}"
            done ;;
        add)
            local name=${1:?a name for the caller, e.g. ./gate keys add thinkera-platform}
            if grep -q "\"$name\";" "$KEYS" 2>/dev/null; then
                die "$name already has a key (./gate keys rm $name first)"
            fi
            # `|| true`: head closes the pipe once it has its 40 characters, tr dies of SIGPIPE, and
            # pipefail + set -e would end the script on a key it had just generated correctly.
            local new; new="tk_$(LC_ALL=C tr -dc 'A-Za-z0-9' < /dev/urandom | head -c 40 || true)"
            printf '"%s"  "%s";\n' "$new" "$name" >> "$KEYS"
            dc exec -T gate nginx -s reload >/dev/null 2>&1 || true
            log "$name may now call the gate"
            echo "$new" ;;
        rm|remove|revoke)
            local name=${1:?which caller, e.g. ./gate keys rm thinkera-platform}
            grep -q "\"$name\";" "$KEYS" || die "no caller called $name in $KEYS"
            sed -i "/\"$name\";\$/d" "$KEYS"
            dc exec -T gate nginx -s reload >/dev/null 2>&1 || true
            log "$name revoked, and the door reloaded" ;;
        *) die "./gate keys [list | add <name> | rm <name>]" ;;
    esac
}

test_call() { # one real request, through the door, with a real key
    local key; key=$(sed -n 's/^"\([^"]*\)".*/\1/p' "$KEYS" | head -1)
    [ -n "$key" ] || die "no key in $KEYS yet (./gate keys add <name>)"
    local model=${1:-}
    [ -n "$model" ] || model=$(dc exec -T llm ollama list 2>/dev/null | awk 'NR==2{print $1}')
    [ -n "$model" ] || die "no model installed yet (./gate pull <model>)"
    log "asking $model through the door…"
    curl -s -m 600 "http://127.0.0.1:$(port)/v1/chat/completions" \
        -H "Authorization: Bearer $key" -H 'Content-Type: application/json' \
        -d "{\"model\":\"$model\",\"messages\":[{\"role\":\"user\",\"content\":\"Reply with the one word: ready\"}],\"max_tokens\":16}"
    echo
}

update() {
    # The models live in ./models, not in the image, so neither of these touches them.
    local was; was=$(dc exec -T llm ollama --version 2>/dev/null | tr -d '\r' | tail -1)
    log "fetching the newest Ollama image"
    dc pull llm
    log "recreating the container"
    dc up -d llm
    wait_up
    local now; now=$(dc exec -T llm ollama --version 2>/dev/null | tr -d '\r' | tail -1)
    log "was: ${was:-unknown}"
    log "now: ${now:-unknown}"
    echo; echo 'models (they live in ./models and are not affected):'
    dc exec -T llm ollama list
}

wait_up() {  # the container answers before it is 'healthy'; wait for the model server, not the healthcheck
    local i=0
    while [ $i -lt 60 ]; do
        dc exec -T llm ollama list >/dev/null 2>&1 && return 0
        sleep 2; i=$((i + 1))
    done
    die "the model container did not come back; ./gate logs llm"
}

cmd=${1:-help}; shift || true
case $cmd in
    up)      first_run; dc up -d "$@"; echo; status ;;
    update)  first_run; update ;;
    down)    dc down ;;
    restart) dc restart "$@" ;;
    status)  status ;;
    logs)    dc logs -f --tail=200 "$@" ;;
    sh)      dc exec "${1:?service: llm or gate}" sh ;;
    pull)
        local_models=${*:-$(envval MODELS)}
        [ -n "$local_models" ] || die "name a model, e.g. ./gate pull qwen3:30b-a3b (or set MODELS in .env)"
        dc up -d llm >/dev/null
        for m in $local_models; do log "pulling $m"; dc exec -T llm ollama pull "$m"; done
        dc exec -T llm ollama list ;;
    list)    dc exec -T llm ollama list ;;
    rm)      dc exec -T llm ollama rm "${1:?which model}" ;;
    keys)    first_run; keys "$@" ;;
    test)    test_call "$@" ;;
    help|*)
        cat <<'EOF'
LLM-Gate — a local language model with a door on it.

  ./gate up                      start it (creates .env and keys.conf on the first run)
  ./gate status                  containers, health, models, callers
  ./gate pull qwen3:30b-a3b      download a model into ./models
  ./gate update                  newer Ollama (a model published after it refuses to pull); keeps ./models
  ./gate list                    which models are installed
  ./gate rm <model>              delete one
  ./gate keys                    who may call
  ./gate keys add <name>         mint a key for a caller and print it once
  ./gate keys rm <name>          take it away (the door reloads immediately)
  ./gate test [model]            one real request through the door, with a real key
  ./gate logs [llm|gate]         follow the logs (the gate's log names each caller)
  ./gate down                    stop and remove the containers (./models stays)

Callers reach it at  http://<this host>:11444  with the key in a header:
  Authorization: Bearer <key>        or        X-API-Key: <key>
OpenAI-compatible clients point their base URL at  http://<this host>:11444/v1/
EOF
        ;;
esac
