Files
llm-server-setup/setup_llm_server.sh

142 lines
4.2 KiB
Bash

#!/usr/bin/env bash
set -e
# Detectar usuario normal si se ejecuta con sudo
TARGET_USER="${SUDO_USER:-$USER}"
TARGET_HOME=$(getent passwd "$TARGET_USER" | cut -d: -f6)
echo "=== [1/8] Actualizando paquetes base del sistema ==="
export DEBIAN_FRONTEND=noninteractive
apt-get update && apt-get upgrade -y
apt-get install -y curl wget git build-essential ufw jq python3-pip python3-venv smartmontools
echo "=== [2/8] Configurando Swap de 8 GB ==="
if [ ! -f /swapfile ]; then
fallocate -l 8G /swapfile || dd if=/dev/zero of=/swapfile bs=1M count=8192
chmod 600 /swapfile
mkswap /swapfile
swapon /swapfile
if ! grep -q '/swapfile' /etc/fstab; then
echo '/swapfile none swap sw 0 0' >> /etc/fstab
fi
sysctl vm.swappiness=10
echo 'vm.swappiness=10' > /etc/sysctl.d/99-swappiness.conf
fi
echo "=== [3/8] Instalando Drivers NVIDIA y CUDA ==="
if ! command -v nvidia-smi &> /dev/null; then
apt-get install -y ubuntu-drivers-common
ubuntu-drivers install --gpgpu || apt-get install -y nvidia-driver-535-server
fi
echo "=== [4/8] Configurando servicio systemd para Power Limit (300W) ==="
cat << 'EOF' > /etc/systemd/system/nvidia-power-limit.service
[Unit]
Description=Set NVIDIA GPU Power Limit to 300W
After=systemd-modules-load.service
[Service]
Type=oneshot
ExecStart=/usr/bin/nvidia-smi -pm 1
ExecStart=/usr/bin/nvidia-smi -pl 300
RemainAfterExit=yes
[Install]
WantedBy=multi-user.target
EOF
systemctl daemon-reload
systemctl enable nvidia-power-limit.service
# Intentar aplicar de inmediato si el driver ya está cargado
nvidia-smi -pm 1 2>/dev/null || true
nvidia-smi -pl 300 2>/dev/null || true
echo "=== [5/8] Instalando Ollama y configurando servicio ==="
if ! command -v ollama &> /dev/null; then
curl -fsSL https://ollama.com/install.sh | sh
fi
# Configurar variables de Ollama para acceso de red y precarga
mkdir -p /etc/systemd/system/ollama.service.d
cat << 'EOF' > /etc/systemd/system/ollama.service.d/override.conf
[Service]
Environment="OLLAMA_HOST=0.0.0.0:11434"
Environment="OLLAMA_KEEP_ALIVE=-1"
Environment="OLLAMA_NUM_PARALLEL=1"
EOF
systemctl daemon-reload
systemctl restart ollama
systemctl enable ollama
echo "=== [6/8] Configurando LiteLLM Proxy en entorno virtual dedicado ==="
rm -rf /opt/litellm-env
python3 -m venv /opt/litellm-env
/opt/litellm-env/bin/pip install --upgrade pip
/opt/litellm-env/bin/pip install 'litellm[proxy]' nvitop
# Crear configuración de LiteLLM para Tool Calling y VS Code
cat << 'EOF' > "$TARGET_HOME/litellm_config.yaml"
model_list:
- model_name: qwen2.5-coder:14b
litellm_params:
model: ollama/qwen2.5-coder:14b
api_base: http://127.0.0.1:11434
max_tokens: 32768
litellm_settings:
drop_params: true
set_verbose: false
EOF
chown "$TARGET_USER:$TARGET_USER" "$TARGET_HOME/litellm_config.yaml"
# Crear servicio systemd para LiteLLM
cat << EOF > /etc/systemd/system/litellm.service
[Unit]
Description=LiteLLM Proxy Service for VS Code Agent
After=network.target ollama.service
Wants=ollama.service
[Service]
Type=simple
User=root
WorkingDirectory=$TARGET_HOME
ExecStart=/opt/litellm-env/bin/litellm --config $TARGET_HOME/litellm_config.yaml --port 8000 --host 0.0.0.0
Restart=always
RestartSec=5
Environment=PYTHONUNBUFFERED=1
[Install]
WantedBy=multi-user.target
EOF
systemctl daemon-reload
systemctl enable litellm.service
echo "=== [7/8] Configurando Firewall (UFW) ==="
ufw allow 22/tcp comment 'SSH'
ufw allow 8000/tcp comment 'LiteLLM Proxy'
ufw --force enable
echo "=== [8/8] Descargando modelo qwen2.5-coder:14b ==="
# Esperar que Ollama responda antes del pull
until curl -s http://127.0.0.1:11434/api/tags > /dev/null; do
echo "Esperando a que Ollama levante..."
sleep 2
done
ollama pull qwen2.5-coder:14b
# Iniciar LiteLLM ahora que el modelo existe
systemctl start litellm.service
echo ""
echo "=========================================================="
echo " ¡INSTALACIÓN COMPLETADA EXITOSAMENTE!"
echo "=========================================================="
echo "IP del Servidor: $(hostname -I | awk '{print $1}')"
echo "Endpoint LiteLLM: http://$(hostname -I | awk '{print $1}'):8000"
echo ""
echo "Nota: Si los drivers NVIDIA eran nuevos, se recomienda un reinicio:"
echo "sudo reboot"
echo "=========================================================="