#!/usr/bin/env bash
#
# Inferix GPU Host setup
# Usage:
#   curl -fsSL https://inferix.co/install-provider.sh | bash -s -- --api-key YOUR_API_KEY
#
# Optional flags:
#   --price     <usd_per_gpu_hour>  Listing price per GPU/hour (default 0.50)
#   --region    <region>            e.g. us-east (default: unknown)
#   --country   <code>              e.g. US (default: XX)
#   --api       <base_url>          API base (default https://inferix.co/api/v0)
#   --interval  <seconds>           Heartbeat interval (default 60)
#   --agent-url <url>               Override agent download URL
#   --daemon                        Force-install the heartbeat daemon
#   --no-daemon                     Register only; skip the daemon
#
# This registers your machine's hardware and (by default) installs a small
# heartbeat agent as a systemd service. Your machine shows "online" in the
# marketplace only while that agent is actively reporting.

set -euo pipefail

API_BASE="https://inferix.co/api/v0"
API_KEY=""
PRICE="0.50"
REGION="unknown"
COUNTRY="XX"
INTERVAL="60"
AGENT_URL=""
INSTALL_DAEMON="auto"   # auto | yes | no

while [ "$#" -gt 0 ]; do
  case "$1" in
    --api-key)   API_KEY="${2:-}"; shift 2 ;;
    --price)     PRICE="${2:-}"; shift 2 ;;
    --region)    REGION="${2:-}"; shift 2 ;;
    --country)   COUNTRY="${2:-}"; shift 2 ;;
    --api)       API_BASE="${2:-}"; shift 2 ;;
    --interval)  INTERVAL="${2:-}"; shift 2 ;;
    --agent-url) AGENT_URL="${2:-}"; shift 2 ;;
    --daemon)    INSTALL_DAEMON="yes"; shift 1 ;;
    --no-daemon) INSTALL_DAEMON="no"; shift 1 ;;
    *) echo "Unknown argument: $1" >&2; exit 1 ;;
  esac
done

if [ -z "$API_KEY" ]; then
  echo "Error: --api-key is required." >&2
  echo "Create one at https://inferix.co/settings/tokens then re-run:" >&2
  echo "  curl -fsSL https://inferix.co/install/provider | bash -s -- --api-key YOUR_API_KEY" >&2
  exit 1
fi

echo "[1/4] Fingerprinting hardware (read-only)..."

# --- GPU ---
GPU_TYPE="CPU-only"
GPU_COUNT=0
GPU_MEM_MIB=0
if command -v nvidia-smi >/dev/null 2>&1; then
  GPU_TYPE="$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | head -1 | sed 's/^ *//;s/ *$//')"
  GPU_COUNT="$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | wc -l | tr -d ' ')"
  GPU_MEM_MIB="$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d ' ')"
fi
[ -z "$GPU_TYPE" ] && GPU_TYPE="Unknown GPU"
[ -z "$GPU_COUNT" ] && GPU_COUNT=0
[ -z "$GPU_MEM_MIB" ] && GPU_MEM_MIB=0

if [ "$GPU_COUNT" -lt 1 ]; then
  echo "Warning: no NVIDIA GPU detected. Registering as a CPU-only host." >&2
  GPU_COUNT=1
fi

# --- CPU ---
CPU_MODEL="$(LC_ALL=C lscpu 2>/dev/null | sed -n 's/^Model name:[[:space:]]*//p' | head -1)"
[ -z "$CPU_MODEL" ] && CPU_MODEL="Unknown CPU"
CPU_CORES="$(nproc 2>/dev/null || echo 1)"

# --- RAM ---
RAM_KB="$(grep -m1 MemTotal /proc/meminfo 2>/dev/null | awk '{print $2}')"
[ -z "$RAM_KB" ] && RAM_KB=1048576
RAM_GB="$(( RAM_KB / 1048576 ))"
[ "$RAM_GB" -lt 1 ] && RAM_GB=1

# --- Disk ---
DISK_GB="$(df -BG --output=avail / 2>/dev/null | tail -1 | tr -dc '0-9')"
[ -z "$DISK_GB" ] && DISK_GB=1

# --- Stable machine fingerprint id (no raw serials leave the host) ---
RAW_ID="$( (cat /etc/machine-id 2>/dev/null; cat /sys/class/dmi/id/product_uuid 2>/dev/null; hostname) | tr -d '\n' )"
if command -v sha256sum >/dev/null 2>&1; then
  MACHINE_ID="ifx-$(printf '%s' "$RAW_ID" | sha256sum | cut -c1-32)"
else
  MACHINE_ID="ifx-$(printf '%s' "$RAW_ID" | cksum | tr -d ' ' )"
fi

echo "    GPU:    $GPU_COUNT x $GPU_TYPE (${GPU_MEM_MIB} MiB)"
echo "    CPU:    $CPU_MODEL ($CPU_CORES cores)"
echo "    RAM:    ${RAM_GB} GB    Disk: ${DISK_GB} GB"
echo "    Machine id: $MACHINE_ID"

echo "[2/4] Registering machine with Inferix..."

# JSON-escape the few free-text fields.
json_escape() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; }

PAYLOAD="$(cat <<JSON
{
  "machineId": "$(json_escape "$MACHINE_ID")",
  "gpuType": "$(json_escape "$GPU_TYPE")",
  "gpuCount": $GPU_COUNT,
  "gpuMemory": $GPU_MEM_MIB,
  "cpuModel": "$(json_escape "$CPU_MODEL")",
  "cpuCores": $CPU_CORES,
  "ramGb": $RAM_GB,
  "storageGb": $DISK_GB,
  "storageType": "NVMe/SSD",
  "networkSpeed": "unknown",
  "region": "$(json_escape "$REGION")",
  "country": "$(json_escape "$COUNTRY")",
  "pricePerGpuPerHour": $PRICE,
  "agentVersion": "phase2-installer"
}
JSON
)"

HTTP_CODE="$(curl -sS -o /tmp/inferix_register_resp.json -w '%{http_code}' \
  -X POST "$API_BASE/machines/agent/register" \
  -H "Authorization: Bearer $API_KEY" \
  -H "Content-Type: application/json" \
  --data "$PAYLOAD" || echo "000")"

if [ "$HTTP_CODE" != "200" ] && [ "$HTTP_CODE" != "201" ]; then
  echo "Registration failed (HTTP $HTTP_CODE):" >&2
  cat /tmp/inferix_register_resp.json 2>/dev/null >&2 || true
  echo >&2
  echo "Check that your API key is valid (https://inferix.co/settings/tokens)." >&2
  exit 1
fi

echo "[3/4] Installing heartbeat agent..."

# Where the agent script is served (strip the /api/v0 suffix from the API base).
if [ -z "$AGENT_URL" ]; then
  AGENT_URL="$(printf '%s' "$API_BASE" | sed -E 's#/api/v0/?$##')/inferix-agent.sh"
fi

# Decide whether we can install the systemd daemon.
SUDO=""
if [ "$(id -u)" -ne 0 ]; then
  if command -v sudo >/dev/null 2>&1; then SUDO="sudo"; fi
fi

has_systemd() {
  command -v systemctl >/dev/null 2>&1 && [ -d /run/systemd/system ]
}
can_root() {
  [ "$(id -u)" -eq 0 ] || [ -n "$SUDO" ]
}

DAEMON_INSTALLED="no"
if [ "$INSTALL_DAEMON" = "no" ]; then
  echo "    Skipping daemon install (--no-daemon)."
elif ! has_systemd || ! can_root; then
  if [ "$INSTALL_DAEMON" = "yes" ]; then
    echo "    Warning: cannot install systemd service (need root + systemd). Falling back to manual instructions." >&2
  else
    echo "    systemd/root not available — will print manual run instructions instead."
  fi
else
  # 1) Config file holding the agent's credentials + settings.
  $SUDO mkdir -p /etc/inferix
  $SUDO tee /etc/inferix/agent.conf >/dev/null <<CONF
API_BASE="$API_BASE"
API_KEY="$API_KEY"
MACHINE_ID="$MACHINE_ID"
INTERVAL="$INTERVAL"
CONF
  $SUDO chmod 600 /etc/inferix/agent.conf

  # 2) Download + install the agent binary.
  TMP_AGENT="$(mktemp)"
  if curl -fsSL "$AGENT_URL" -o "$TMP_AGENT" && [ -s "$TMP_AGENT" ]; then
    $SUDO install -m 0755 "$TMP_AGENT" /usr/local/bin/inferix-agent
    rm -f "$TMP_AGENT"

    # 3) systemd unit (auto-restart, starts on boot).
    $SUDO tee /etc/systemd/system/inferix-agent.service >/dev/null <<'UNIT'
[Unit]
Description=Inferix GPU host agent (heartbeat)
After=network-online.target
Wants=network-online.target

[Service]
Type=simple
EnvironmentFile=/etc/inferix/agent.conf
ExecStart=/usr/local/bin/inferix-agent
Restart=always
RestartSec=10
NoNewPrivileges=true
PrivateTmp=true
ProtectSystem=full
ProtectHome=true

[Install]
WantedBy=multi-user.target
UNIT

    $SUDO systemctl daemon-reload
    if $SUDO systemctl enable --now inferix-agent.service >/dev/null 2>&1; then
      DAEMON_INSTALLED="yes"
      echo "    Agent installed and started (systemd: inferix-agent.service)."
    else
      echo "    Warning: agent installed but failed to start. Check: $SUDO systemctl status inferix-agent" >&2
    fi
  else
    rm -f "$TMP_AGENT"
    echo "    Warning: could not download the agent from $AGENT_URL" >&2
  fi
fi

##############################################################################
# [4/5] Provisioning agent — this is what actually makes the machine RENTABLE.
#
# The heartbeat agent above only reports "online". Without the provisioning
# agent the backend's isAgentBacked() check is false, so the listing can NEVER
# be rented: no /instances endpoint, no podman, no GPU containers. Previously
# this installer stopped at the heartbeat, which is why new hosts appeared
# online but never received work.
##############################################################################
echo "[4/5] Installing provisioning agent (makes the machine rentable)..."

BASE_URL="$(printf '%s' "$API_BASE" | sed -E 's#/api/v0/?$##')"
PROV_OK="no"

if [ "$INSTALL_DAEMON" = "no" ]; then
  echo "    Skipped (--no-daemon)."
elif ! has_systemd || ! can_root; then
  echo "    Skipped: needs root + systemd. Machine shows online but is NOT rentable." >&2
elif ! command -v nvidia-smi >/dev/null 2>&1; then
  echo "    Skipped: nvidia-smi not found. Install the NVIDIA driver, then re-run." >&2
else
  # ---- dependencies: podman + NVIDIA CDI ----------------------------------
  if ! command -v podman >/dev/null 2>&1; then
    echo "    Installing podman..."
    if command -v apt-get >/dev/null 2>&1; then
      $SUDO apt-get update -qq >/dev/null 2>&1 && \
        $SUDO DEBIAN_FRONTEND=noninteractive apt-get install -y -qq podman >/dev/null 2>&1
    elif command -v dnf >/dev/null 2>&1; then
      $SUDO dnf install -y -q podman >/dev/null 2>&1
    else
      echo "    Warning: no apt/dnf found — install podman manually." >&2
    fi
  fi

  # CDI lets podman expose GPUs via --device nvidia.com/gpu=N (no --privileged).
  if command -v nvidia-ctk >/dev/null 2>&1; then
    $SUDO mkdir -p /etc/cdi
    $SUDO nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml >/dev/null 2>&1 \
      || echo "    Warning: CDI generation failed; GPU passthrough may not work." >&2
  else
    echo "    Warning: nvidia-ctk (NVIDIA Container Toolkit) not found." >&2
    echo "             Install it, then: sudo nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml" >&2
  fi

  # ---- base image: built locally (it is NOT published to any registry) ----
  if ! $SUDO podman image exists localhost/inferix-gpu-base:latest 2>/dev/null; then
    echo "    Building the renter base image (several minutes, one time)..."
    BI_DIR="$(mktemp -d)"
    if curl -fsSL "$BASE_URL/gpu-agent/base-image/Dockerfile" -o "$BI_DIR/Dockerfile" \
       && curl -fsSL "$BASE_URL/gpu-agent/base-image/entrypoint.sh" -o "$BI_DIR/entrypoint.sh"; then
      chmod +x "$BI_DIR/entrypoint.sh"
      if $SUDO podman build -t inferix-gpu-base:latest "$BI_DIR" >/dev/null 2>&1; then
        echo "    Base image built."
      else
        echo "    Warning: base image build failed; instances will not start." >&2
      fi
    else
      echo "    Warning: could not download base-image sources from $BASE_URL." >&2
    fi
    rm -rf "$BI_DIR"
  fi

  # ---- the provisioning agent itself --------------------------------------
  TMP_PROV="$(mktemp)"
  if curl -fsSL "$BASE_URL/gpu-agent/agent.py" -o "$TMP_PROV" && [ -s "$TMP_PROV" ]; then
    $SUDO mkdir -p /opt/inferix-gpu-agent /var/lib/inferix-gpu-agent
    $SUDO install -m 0755 "$TMP_PROV" /opt/inferix-gpu-agent/agent.py
    rm -f "$TMP_PROV"

    PROV_TOKEN="$(head -c 32 /dev/urandom | od -An -tx1 | tr -d ' \n')"
    GPU_LIST="$(nvidia-smi --query-gpu=index --format=csv,noheader 2>/dev/null | paste -sd, -)"
    GPU_NAME_DETECTED="$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | head -1)"
    GPU_MEM_MB="$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null | head -1)"
    HOST_IP="$(ip route get 1.1.1.1 2>/dev/null | awk '/src/ {print $7; exit}')"

    $SUDO tee /etc/inferix-gpu-agent.env >/dev/null <<ENVEOF
AGENT_TOKEN=$PROV_TOKEN
AGENT_PORT=8102
AGENT_HOST_IP=${HOST_IP:-127.0.0.1}
RENTABLE_GPUS=${GPU_LIST:-0}
PORT_RANGE_START=42000
PORT_RANGE_END=42099
BASE_IMAGE=localhost/inferix-gpu-base:latest
SERVE_IMAGE=ghcr.io/ggml-org/llama.cpp:server-cuda
STATE_FILE=/var/lib/inferix-gpu-agent/state.json
INFERIX_API_URL=$API_BASE
INFERIX_API_TOKEN=$API_KEY
MACHINE_ID=$MACHINE_ID
GPU_TYPE=${GPU_NAME_DETECTED:-Unknown GPU}
GPU_MEMORY_MB=${GPU_MEM_MB:-0}
REGION=${REGION:-us-west}
COUNTRY=${COUNTRY:-US}
HEARTBEAT_INTERVAL=30
PUBLIC_PROXY_HOST=inferix.co
ENVEOF
    $SUDO chmod 600 /etc/inferix-gpu-agent.env

    $SUDO tee /etc/systemd/system/inferix-gpu-agent.service >/dev/null <<'PUNIT'
[Unit]
Description=Inferix GPU Host Agent (real podman GPU container provisioning)
After=network-online.target
Wants=network-online.target

[Service]
Type=simple
EnvironmentFile=/etc/inferix-gpu-agent.env
ExecStart=/usr/bin/python3 /opt/inferix-gpu-agent/agent.py
Restart=always
RestartSec=10

[Install]
WantedBy=multi-user.target
PUNIT

    # Renter containers get their OWN network namespace with only their two rented
    # ports published. ufw's FORWARD policy is DROP and podman installs no FORWARD
    # rules, so without this the published ports are unreachable and the public
    # Jupyter proxy breaks. Scoped: LAN -> podman subnet only.
    #
    # That same DROP policy also blocks the containers' OUTBOUND traffic, which
    # left renters on an instance with no internet at all: no pip install, no git
    # clone, no downloading weights or datasets. The rules below restore egress
    # while keeping the provider's own network off-limits.
    #
    # Order matters — ufw evaluates top-down and first match wins, so every
    # private-range DENY must be added BEFORE the catch-all ALLOW. Inbound
    # Jupyter/SSH is unaffected: ufw-before-forward accepts RELATED,ESTABLISHED
    # ahead of these user rules, so replies on an established connection are not
    # matched by the denies.
    if command -v ufw >/dev/null 2>&1; then
      $SUDO ufw route allow from 10.0.0.0/24 to 10.88.0.0/16 \
        comment 'Inferix: LAN to rented-instance containers' >/dev/null 2>&1 || true

      for PRIVATE_NET in 10.0.0.0/8 192.168.0.0/16 172.16.0.0/12 169.254.0.0/16; do
        $SUDO ufw route deny from 10.88.0.0/16 to "$PRIVATE_NET" \
          comment 'Inferix: renters must not reach the provider network' >/dev/null 2>&1 || true
      done

      $SUDO ufw route allow from 10.88.0.0/16 to any \
        comment 'Inferix: renter containers need internet (pip/git/datasets)' >/dev/null 2>&1 || true
    fi

    $SUDO systemctl daemon-reload
    if $SUDO systemctl enable --now inferix-gpu-agent.service >/dev/null 2>&1; then
      sleep 3
      if curl -fsS --max-time 8 "http://127.0.0.1:8102/health" >/dev/null 2>&1; then
        PROV_OK="yes"
        echo "    Provisioning agent healthy — this machine is RENTABLE."
      else
        echo "    Warning: agent started but /health did not respond." >&2
        echo "             Check: $SUDO journalctl -u inferix-gpu-agent -n 50" >&2
      fi
    else
      echo "    Warning: provisioning agent failed to start." >&2
    fi
  else
    rm -f "$TMP_PROV"
    echo "    Warning: could not download the provisioning agent from $BASE_URL/gpu-agent/agent.py" >&2
  fi
fi

echo "[5/5] Done."
echo
if [ "$PROV_OK" != "yes" ]; then
  echo " NOTE: the provisioning agent is NOT running, so this machine will show"
  echo "       online but will not receive rentals. Re-run as root after"
  echo "       installing the NVIDIA driver + Container Toolkit."
  echo
fi
echo "========================================================"
if [ "$DAEMON_INSTALLED" = "yes" ]; then
  echo " Your machine is registered and the heartbeat agent is"
  echo " running. It will appear online in the marketplace and"
  echo " auto-verify after a few successful heartbeats."
  echo
  echo " Agent logs:  journalctl -u inferix-agent -f"
  echo " Stop agent:  systemctl stop inferix-agent"
else
  echo " Your machine is registered (pending review)."
  echo
  echo " To keep it online, run the heartbeat agent:"
  echo "   curl -fsSL $AGENT_URL -o inferix-agent.sh && chmod +x inferix-agent.sh"
  echo "   API_BASE=\"$API_BASE\" API_KEY=\"<your-api-key>\" \\"
  echo "     MACHINE_ID=\"$MACHINE_ID\" ./inferix-agent.sh &"
fi
echo "========================================================"
echo " Verify it:   https://inferix.co/cloud/host/setup  (Verify my machine)"
echo " Manage it:   https://inferix.co/cloud/host/machines"
echo " Set pricing: https://inferix.co/provider/dashboard"
echo
echo " Renters launch GPU-pinned containers on this host with SSH,"
echo " Jupyter, persistent volumes, and OpenAI-compatible inference."
echo " For full workload orchestration, ensure podman + the NVIDIA"
echo " Container Toolkit (CDI) are installed so instances can start."
