534 lines
12 KiB
Bash
534 lines
12 KiB
Bash
#!/bin/bash
|
|
|
|
################################################################################
|
|
# LLM SERVER DEPLOYMENT SCRIPT
|
|
# Ubuntu 22.04 LTS + ASUS H510M + RTX 3090
|
|
#
|
|
# Uso: bash scripts/install.sh [--quick] [--gpu-only]
|
|
#
|
|
# Opciones:
|
|
# --quick Salta verificaciones lentas
|
|
# --gpu-only Solo instala GPU drivers (para re-install)
|
|
################################################################################
|
|
|
|
set -e # Exit si hay error
|
|
|
|
# Colores
|
|
RED='\033[0;31m'
|
|
GREEN='\033[0;32m'
|
|
YELLOW='\033[1;33m'
|
|
BLUE='\033[0;34m'
|
|
NC='\033[0m' # No Color
|
|
|
|
# Variables
|
|
SCRIPT_DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )"
|
|
PROJECT_DIR="$(dirname "$SCRIPT_DIR")"
|
|
WORKSPACE_DIR="$PROJECT_DIR/workspace"
|
|
VENV_DIR="$PROJECT_DIR/venv"
|
|
LOG_FILE="$PROJECT_DIR/logs/install.log"
|
|
|
|
# Crear directorio de logs
|
|
mkdir -p "$PROJECT_DIR/logs"
|
|
|
|
# Function: Log con timestamp
|
|
log() {
|
|
local level=$1
|
|
shift
|
|
local message="$@"
|
|
local timestamp=$(date '+%Y-%m-%d %H:%M:%S')
|
|
echo -e "${BLUE}[$timestamp]${NC} ${level}: ${message}" | tee -a "$LOG_FILE"
|
|
}
|
|
|
|
# Function: Log success
|
|
log_success() {
|
|
echo -e "${GREEN}✅ $@${NC}" | tee -a "$LOG_FILE"
|
|
}
|
|
|
|
# Function: Log error
|
|
log_error() {
|
|
echo -e "${RED}❌ $@${NC}" | tee -a "$LOG_FILE"
|
|
}
|
|
|
|
# Function: Log warning
|
|
log_warning() {
|
|
echo -e "${YELLOW}⚠️ $@${NC}" | tee -a "$LOG_FILE"
|
|
}
|
|
|
|
# Function: Check command exists
|
|
command_exists() {
|
|
command -v "$1" >/dev/null 2>&1
|
|
}
|
|
|
|
# Function: Check if running as root
|
|
check_root() {
|
|
if [[ $EUID -ne 0 ]]; then
|
|
log_error "Este script debe ejecutarse con sudo"
|
|
exit 1
|
|
fi
|
|
}
|
|
|
|
################################################################################
|
|
# MAIN INSTALLATION
|
|
################################################################################
|
|
|
|
main() {
|
|
local quick_mode=false
|
|
local gpu_only=false
|
|
|
|
# Parse arguments
|
|
while [[ $# -gt 0 ]]; do
|
|
case $1 in
|
|
--quick)
|
|
quick_mode=true
|
|
shift
|
|
;;
|
|
--gpu-only)
|
|
gpu_only=true
|
|
shift
|
|
;;
|
|
*)
|
|
log_error "Opción desconocida: $1"
|
|
usage
|
|
exit 1
|
|
;;
|
|
esac
|
|
done
|
|
|
|
log "LOG" "=========================================="
|
|
log "LOG" "LLM Server Installation"
|
|
log "LOG" "=========================================="
|
|
log "LOG" "Project Dir: $PROJECT_DIR"
|
|
log "LOG" "Workspace: $WORKSPACE_DIR"
|
|
log "LOG" "Quick Mode: $quick_mode"
|
|
log "LOG" "GPU Only: $gpu_only"
|
|
log "LOG" "=========================================="
|
|
|
|
# Checks iniciales
|
|
check_root
|
|
check_os
|
|
check_hardware
|
|
|
|
if [ "$gpu_only" = false ]; then
|
|
install_dependencies
|
|
install_docker
|
|
fi
|
|
|
|
install_nvidia_drivers
|
|
install_cuda_toolkit
|
|
install_ollama
|
|
setup_python_venv
|
|
create_services
|
|
|
|
if [ "$quick_mode" = false ]; then
|
|
download_models
|
|
fi
|
|
|
|
setup_directories
|
|
generate_config
|
|
|
|
log_success "=========================================="
|
|
log_success "✨ INSTALACIÓN COMPLETADA"
|
|
log_success "=========================================="
|
|
print_next_steps
|
|
}
|
|
|
|
################################################################################
|
|
# FUNCIONES AUXILIARES
|
|
################################################################################
|
|
|
|
check_os() {
|
|
log "LOG" "Verificando Sistema Operativo..."
|
|
|
|
if [ ! -f /etc/os-release ]; then
|
|
log_error "No se pudo detectar el SO"
|
|
exit 1
|
|
fi
|
|
|
|
. /etc/os-release
|
|
if [[ "$ID" != "ubuntu" ]]; then
|
|
log_error "Este script solo soporta Ubuntu"
|
|
exit 1
|
|
fi
|
|
|
|
if [[ "$VERSION_ID" != "22.04" ]]; then
|
|
log_warning "Se recomienda Ubuntu 22.04, detectado: $VERSION_ID"
|
|
fi
|
|
|
|
log_success "Ubuntu $VERSION_ID detectado"
|
|
}
|
|
|
|
check_hardware() {
|
|
log "LOG" "Verificando Hardware..."
|
|
|
|
# Check GPU
|
|
if ! command_exists nvidia-smi; then
|
|
log_warning "nvidia-smi no disponible aún (se instalará)"
|
|
else
|
|
gpu_info=$(nvidia-smi --query-gpu=name,memory.total --format=csv,noheader)
|
|
log_success "GPU detectada: $gpu_info"
|
|
fi
|
|
|
|
# Check CPU
|
|
cpu_count=$(nproc)
|
|
log_success "CPU: $cpu_count cores"
|
|
|
|
# Check RAM
|
|
ram_gb=$(free -h | awk '/^Mem:/ {print $2}')
|
|
log_success "RAM: $ram_gb"
|
|
|
|
# Check Disk
|
|
disk_info=$(df -h / | awk 'NR==2 {print $2}')
|
|
log_success "Disco: $disk_info disponible"
|
|
}
|
|
|
|
install_dependencies() {
|
|
log "LOG" "Instalando dependencias del sistema..."
|
|
|
|
apt update
|
|
apt install -y \
|
|
build-essential \
|
|
git \
|
|
wget \
|
|
curl \
|
|
htop \
|
|
nano \
|
|
openssh-server \
|
|
python3.11 \
|
|
python3.11-venv \
|
|
python3.11-dev \
|
|
pkg-config \
|
|
libssl-dev \
|
|
libffi-dev
|
|
|
|
log_success "Dependencias instaladas"
|
|
}
|
|
|
|
install_docker() {
|
|
log "LOG" "Instalando Docker..."
|
|
|
|
if command_exists docker; then
|
|
log_success "Docker ya está instalado"
|
|
return
|
|
fi
|
|
|
|
curl -fsSL https://get.docker.com -o /tmp/get-docker.sh
|
|
sh /tmp/get-docker.sh
|
|
|
|
# Agregar usuario al grupo docker
|
|
if id "charle" &>/dev/null; then
|
|
usermod -aG docker charle
|
|
log_success "Usuario 'charle' agregado al grupo docker"
|
|
fi
|
|
|
|
systemctl start docker
|
|
systemctl enable docker
|
|
|
|
log_success "Docker instalado"
|
|
}
|
|
|
|
install_nvidia_drivers() {
|
|
log "LOG" "Instalando NVIDIA Drivers..."
|
|
|
|
if command_exists nvidia-smi; then
|
|
current_driver=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -1)
|
|
log_success "Driver NVIDIA $current_driver ya instalado"
|
|
return
|
|
fi
|
|
|
|
# Agregar repositorio NVIDIA
|
|
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys A4B469963BF863CC 2>&1 | grep -v "Warning" || true
|
|
|
|
add-apt-repository "deb http://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/ /" 2>&1 | grep -v "already" || true
|
|
|
|
apt update
|
|
apt install -y cuda-drivers
|
|
|
|
log_success "NVIDIA Drivers instalados"
|
|
|
|
log_warning "Se recomienda reiniciar: sudo reboot"
|
|
}
|
|
|
|
install_cuda_toolkit() {
|
|
log "LOG" "Instalando CUDA Toolkit 12.3..."
|
|
|
|
if [ -d "/usr/local/cuda-12.3" ]; then
|
|
log_success "CUDA 12.3 ya está instalado"
|
|
return
|
|
fi
|
|
|
|
log "LOG" "Descargando CUDA 12.3.0..."
|
|
cd /tmp
|
|
wget -q https://developer.download.nvidia.com/compute/cuda/12.3.0/local_installers/cuda_12.3.0_545.23.06_linux.run \
|
|
-O cuda_12.3.0_545.23.06_linux.run
|
|
|
|
log "LOG" "Instalando CUDA (esto puede tardar 10-15 minutos)..."
|
|
chmod +x cuda_12.3.0_545.23.06_linux.run
|
|
./cuda_12.3.0_545.23.06_linux.run --silent --driver=false --toolkit
|
|
|
|
# Configurar PATH
|
|
if ! grep -q "cuda-12.3" /root/.bashrc; then
|
|
cat >> /root/.bashrc << 'EOF'
|
|
|
|
# CUDA 12.3
|
|
export PATH=/usr/local/cuda-12.3/bin:$PATH
|
|
export LD_LIBRARY_PATH=/usr/local/cuda-12.3/lib64:$LD_LIBRARY_PATH
|
|
EOF
|
|
fi
|
|
|
|
source /root/.bashrc
|
|
|
|
# Verificar
|
|
if command_exists nvcc; then
|
|
cuda_version=$(nvcc --version | grep "release" | awk '{print $5}')
|
|
log_success "CUDA $cuda_version instalado"
|
|
fi
|
|
|
|
# Install cuDNN
|
|
log "LOG" "Instalando cuDNN..."
|
|
apt install -y libcudnn8
|
|
|
|
log_success "CUDA Toolkit instalado"
|
|
}
|
|
|
|
install_ollama() {
|
|
log "LOG" "Instalando Ollama..."
|
|
|
|
if command_exists ollama; then
|
|
log_success "Ollama ya está instalado"
|
|
else
|
|
curl https://ollama.ai/install.sh | sh
|
|
log_success "Ollama instalado"
|
|
fi
|
|
|
|
# Configurar servicio systemd
|
|
log "LOG" "Configurando servicio Ollama..."
|
|
|
|
mkdir -p /etc/systemd/system
|
|
|
|
cat > /etc/systemd/system/ollama.service << 'EOF'
|
|
[Unit]
|
|
Description=Ollama
|
|
After=network-online.target
|
|
|
|
[Service]
|
|
ExecStart=/usr/local/bin/ollama serve
|
|
User=charle
|
|
Group=charle
|
|
Restart=always
|
|
RestartSec=3
|
|
Environment="OLLAMA_HOST=0.0.0.0:11434"
|
|
Environment="OLLAMA_MODELS=/home/charle/.ollama/models"
|
|
Environment="OLLAMA_NUM_GPU=1"
|
|
|
|
[Install]
|
|
WantedBy=default.target
|
|
EOF
|
|
|
|
systemctl daemon-reload
|
|
systemctl enable ollama
|
|
systemctl restart ollama
|
|
|
|
sleep 2
|
|
|
|
if systemctl is-active --quiet ollama; then
|
|
log_success "Servicio Ollama activo"
|
|
else
|
|
log_error "Error iniciando Ollama"
|
|
fi
|
|
}
|
|
|
|
setup_python_venv() {
|
|
log "LOG" "Creando Python Virtual Environment..."
|
|
|
|
if [ -d "$VENV_DIR" ]; then
|
|
log_success "VEnv ya existe"
|
|
return
|
|
fi
|
|
|
|
python3.11 -m venv "$VENV_DIR"
|
|
|
|
source "$VENV_DIR/bin/activate"
|
|
|
|
pip install --upgrade pip setuptools wheel
|
|
|
|
pip install \
|
|
fastapi \
|
|
uvicorn \
|
|
python-multipart \
|
|
aiofiles \
|
|
pillow \
|
|
python-dotenv \
|
|
requests \
|
|
langchain \
|
|
pydantic
|
|
|
|
log_success "Python VEnv configurado"
|
|
}
|
|
|
|
create_services() {
|
|
log "LOG" "Creando systemd services..."
|
|
|
|
# API Service
|
|
cat > /etc/systemd/system/llm-api.service << EOF
|
|
[Unit]
|
|
Description=LLM API Server
|
|
After=network.target ollama.service
|
|
|
|
[Service]
|
|
Type=simple
|
|
User=charle
|
|
WorkingDirectory=$PROJECT_DIR
|
|
Environment="PATH=$VENV_DIR/bin"
|
|
Environment="OLLAMA_URL=http://localhost:11434"
|
|
ExecStart=$VENV_DIR/bin/python $PROJECT_DIR/backend.py
|
|
Restart=always
|
|
RestartSec=10
|
|
|
|
[Install]
|
|
WantedBy=multi-user.target
|
|
EOF
|
|
|
|
systemctl daemon-reload
|
|
systemctl enable llm-api
|
|
|
|
log_success "Systemd services creados"
|
|
}
|
|
|
|
download_models() {
|
|
log "LOG" "Descargando modelos (esto puede tardar 20-30 minutos)..."
|
|
log_warning "Esto es OPCIONAL. Presiona Ctrl+C para cancelar."
|
|
|
|
sleep 5
|
|
|
|
source "$VENV_DIR/bin/activate"
|
|
|
|
# Esperar a que Ollama esté listo
|
|
for i in {1..30}; do
|
|
if curl -s http://localhost:11434/api/tags > /dev/null; then
|
|
log_success "Ollama está listo"
|
|
break
|
|
fi
|
|
log "LOG" "Esperando Ollama... ($i/30)"
|
|
sleep 2
|
|
done
|
|
|
|
log "LOG" "Descargando Phi..."
|
|
ollama pull phi:latest
|
|
|
|
log "LOG" "Descargando DeepSeek Coder 33B..."
|
|
ollama pull deepseek-coder:33b
|
|
|
|
log_success "Modelos descargados"
|
|
}
|
|
|
|
setup_directories() {
|
|
log "LOG" "Creando directorios..."
|
|
|
|
mkdir -p "$WORKSPACE_DIR"
|
|
mkdir -p "$PROJECT_DIR/uploads"
|
|
mkdir -p "$PROJECT_DIR/logs"
|
|
|
|
# Cambiar permisos
|
|
chown -R charle:charle "$PROJECT_DIR"
|
|
chmod -R 755 "$PROJECT_DIR"
|
|
|
|
log_success "Directorios creados"
|
|
}
|
|
|
|
generate_config() {
|
|
log "LOG" "Generando archivos de configuración..."
|
|
|
|
# .env file
|
|
if [ ! -f "$PROJECT_DIR/.env" ]; then
|
|
cat > "$PROJECT_DIR/.env" << EOF
|
|
# LLM Server Configuration
|
|
LLM_SERVER_URL=http://localhost:8000
|
|
LLM_FAST_MODEL=phi
|
|
LLM_POWER_MODEL=deepseek-coder:33b
|
|
OLLAMA_URL=http://localhost:11434
|
|
WORKSPACE_DIR=$WORKSPACE_DIR
|
|
|
|
# Server
|
|
HOST=0.0.0.0
|
|
API_PORT=8000
|
|
WEB_PORT=5000
|
|
|
|
# Logging
|
|
LOG_LEVEL=INFO
|
|
EOF
|
|
log_success ".env creado"
|
|
fi
|
|
|
|
# Config JSON
|
|
if [ ! -f "$PROJECT_DIR/config/server.json" ]; then
|
|
mkdir -p "$PROJECT_DIR/config"
|
|
cat > "$PROJECT_DIR/config/server.json" << 'EOF'
|
|
{
|
|
"server": {
|
|
"host": "0.0.0.0",
|
|
"api_port": 8000,
|
|
"web_port": 5000,
|
|
"workers": 4
|
|
},
|
|
"models": {
|
|
"fast": "phi",
|
|
"power": "deepseek-coder:33b"
|
|
},
|
|
"gpu": {
|
|
"memory_fraction": 0.9,
|
|
"max_batch_size": 8
|
|
},
|
|
"logging": {
|
|
"level": "INFO",
|
|
"format": "%(asctime)s - %(name)s - %(levelname)s - %(message)s"
|
|
}
|
|
}
|
|
EOF
|
|
log_success "config/server.json creado"
|
|
fi
|
|
}
|
|
|
|
print_next_steps() {
|
|
cat << EOF
|
|
|
|
${BLUE}=========================================${NC}
|
|
${GREEN}✨ PRÓXIMOS PASOS:${NC}
|
|
${BLUE}=========================================${NC}
|
|
|
|
1. ${YELLOW}Verificar instalación:${NC}
|
|
sudo systemctl status ollama
|
|
sudo systemctl status llm-api
|
|
|
|
2. ${YELLOW}Ver logs:${NC}
|
|
sudo journalctl -u ollama -f
|
|
sudo journalctl -u llm-api -f
|
|
|
|
3. ${YELLOW}Iniciar servicios:${NC}
|
|
sudo systemctl start ollama
|
|
sudo systemctl start llm-api
|
|
|
|
4. ${YELLOW}Acceder a la API:${NC}
|
|
curl http://localhost:8000/health
|
|
|
|
5. ${YELLOW}Descargar modelos (si no se descargaron):${NC}
|
|
source $VENV_DIR/bin/activate
|
|
ollama pull phi:latest
|
|
ollama pull deepseek-coder:33b
|
|
|
|
6. ${YELLOW}Ver estado en tiempo real:${NC}
|
|
watch -n 1 nvidia-smi
|
|
|
|
${BLUE}=========================================${NC}
|
|
${GREEN}📁 Archivos importantes:${NC}
|
|
${BLUE}=========================================${NC}
|
|
Config: $PROJECT_DIR/config/server.json
|
|
.env: $PROJECT_DIR/.env
|
|
Logs: $PROJECT_DIR/logs/
|
|
Workspace: $WORKSPACE_DIR/
|
|
|
|
${BLUE}=========================================${NC}
|
|
EOF
|
|
}
|
|
|
|
# Ejecutar main
|
|
main "$@" |