#!/bin/bash ################################################################################ # LLM SERVER DEPLOYMENT SCRIPT # Ubuntu 22.04 LTS + ASUS H510M + RTX 3090 # # Uso: bash scripts/install.sh [--quick] [--gpu-only] # # Opciones: # --quick Salta verificaciones lentas # --gpu-only Solo instala GPU drivers (para re-install) ################################################################################ set -e # Exit si hay error # Colores RED='\033[0;31m' GREEN='\033[0;32m' YELLOW='\033[1;33m' BLUE='\033[0;34m' NC='\033[0m' # No Color # Variables SCRIPT_DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )" PROJECT_DIR="$(dirname "$SCRIPT_DIR")" WORKSPACE_DIR="$PROJECT_DIR/workspace" VENV_DIR="$PROJECT_DIR/venv" LOG_FILE="$PROJECT_DIR/logs/install.log" # Crear directorio de logs mkdir -p "$PROJECT_DIR/logs" # Function: Log con timestamp log() { local level=$1 shift local message="$@" local timestamp=$(date '+%Y-%m-%d %H:%M:%S') echo -e "${BLUE}[$timestamp]${NC} ${level}: ${message}" | tee -a "$LOG_FILE" } # Function: Log success log_success() { echo -e "${GREEN}✅ $@${NC}" | tee -a "$LOG_FILE" } # Function: Log error log_error() { echo -e "${RED}❌ $@${NC}" | tee -a "$LOG_FILE" } # Function: Log warning log_warning() { echo -e "${YELLOW}⚠️ $@${NC}" | tee -a "$LOG_FILE" } # Function: Check command exists command_exists() { command -v "$1" >/dev/null 2>&1 } # Function: Check if running as root check_root() { if [[ $EUID -ne 0 ]]; then log_error "Este script debe ejecutarse con sudo" exit 1 fi } ################################################################################ # MAIN INSTALLATION ################################################################################ main() { local quick_mode=false local gpu_only=false # Parse arguments while [[ $# -gt 0 ]]; do case $1 in --quick) quick_mode=true shift ;; --gpu-only) gpu_only=true shift ;; *) log_error "Opción desconocida: $1" usage exit 1 ;; esac done log "LOG" "==========================================" log "LOG" "LLM Server Installation" log "LOG" "==========================================" log "LOG" "Project Dir: $PROJECT_DIR" log "LOG" "Workspace: $WORKSPACE_DIR" log "LOG" "Quick Mode: $quick_mode" log "LOG" "GPU Only: $gpu_only" log "LOG" "==========================================" # Checks iniciales check_root check_os check_hardware if [ "$gpu_only" = false ]; then install_dependencies install_docker fi install_nvidia_drivers install_cuda_toolkit install_ollama setup_python_venv create_services if [ "$quick_mode" = false ]; then download_models fi setup_directories generate_config log_success "==========================================" log_success "✨ INSTALACIÓN COMPLETADA" log_success "==========================================" print_next_steps } ################################################################################ # FUNCIONES AUXILIARES ################################################################################ check_os() { log "LOG" "Verificando Sistema Operativo..." if [ ! -f /etc/os-release ]; then log_error "No se pudo detectar el SO" exit 1 fi . /etc/os-release if [[ "$ID" != "ubuntu" ]]; then log_error "Este script solo soporta Ubuntu" exit 1 fi if [[ "$VERSION_ID" != "22.04" ]]; then log_warning "Se recomienda Ubuntu 22.04, detectado: $VERSION_ID" fi log_success "Ubuntu $VERSION_ID detectado" } check_hardware() { log "LOG" "Verificando Hardware..." # Check GPU if ! command_exists nvidia-smi; then log_warning "nvidia-smi no disponible aún (se instalará)" else gpu_info=$(nvidia-smi --query-gpu=name,memory.total --format=csv,noheader) log_success "GPU detectada: $gpu_info" fi # Check CPU cpu_count=$(nproc) log_success "CPU: $cpu_count cores" # Check RAM ram_gb=$(free -h | awk '/^Mem:/ {print $2}') log_success "RAM: $ram_gb" # Check Disk disk_info=$(df -h / | awk 'NR==2 {print $2}') log_success "Disco: $disk_info disponible" } install_dependencies() { log "LOG" "Instalando dependencias del sistema..." apt update apt install -y \ build-essential \ git \ wget \ curl \ htop \ nano \ openssh-server \ python3.11 \ python3.11-venv \ python3.11-dev \ pkg-config \ libssl-dev \ libffi-dev log_success "Dependencias instaladas" } install_docker() { log "LOG" "Instalando Docker..." if command_exists docker; then log_success "Docker ya está instalado" return fi curl -fsSL https://get.docker.com -o /tmp/get-docker.sh sh /tmp/get-docker.sh # Agregar usuario al grupo docker if id "charle" &>/dev/null; then usermod -aG docker charle log_success "Usuario 'charle' agregado al grupo docker" fi systemctl start docker systemctl enable docker log_success "Docker instalado" } install_nvidia_drivers() { log "LOG" "Instalando NVIDIA Drivers..." if command_exists nvidia-smi; then current_driver=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -1) log_success "Driver NVIDIA $current_driver ya instalado" return fi # Agregar repositorio NVIDIA apt-key adv --keyserver keyserver.ubuntu.com --recv-keys A4B469963BF863CC 2>&1 | grep -v "Warning" || true add-apt-repository "deb http://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/ /" 2>&1 | grep -v "already" || true apt update apt install -y cuda-drivers log_success "NVIDIA Drivers instalados" log_warning "Se recomienda reiniciar: sudo reboot" } install_cuda_toolkit() { log "LOG" "Instalando CUDA Toolkit 12.3..." if [ -d "/usr/local/cuda-12.3" ]; then log_success "CUDA 12.3 ya está instalado" return fi log "LOG" "Descargando CUDA 12.3.0..." cd /tmp wget -q https://developer.download.nvidia.com/compute/cuda/12.3.0/local_installers/cuda_12.3.0_545.23.06_linux.run \ -O cuda_12.3.0_545.23.06_linux.run log "LOG" "Instalando CUDA (esto puede tardar 10-15 minutos)..." chmod +x cuda_12.3.0_545.23.06_linux.run ./cuda_12.3.0_545.23.06_linux.run --silent --driver=false --toolkit # Configurar PATH if ! grep -q "cuda-12.3" /root/.bashrc; then cat >> /root/.bashrc << 'EOF' # CUDA 12.3 export PATH=/usr/local/cuda-12.3/bin:$PATH export LD_LIBRARY_PATH=/usr/local/cuda-12.3/lib64:$LD_LIBRARY_PATH EOF fi source /root/.bashrc # Verificar if command_exists nvcc; then cuda_version=$(nvcc --version | grep "release" | awk '{print $5}') log_success "CUDA $cuda_version instalado" fi # Install cuDNN log "LOG" "Instalando cuDNN..." apt install -y libcudnn8 log_success "CUDA Toolkit instalado" } install_ollama() { log "LOG" "Instalando Ollama..." if command_exists ollama; then log_success "Ollama ya está instalado" else curl https://ollama.ai/install.sh | sh log_success "Ollama instalado" fi # Configurar servicio systemd log "LOG" "Configurando servicio Ollama..." mkdir -p /etc/systemd/system cat > /etc/systemd/system/ollama.service << 'EOF' [Unit] Description=Ollama After=network-online.target [Service] ExecStart=/usr/local/bin/ollama serve User=charle Group=charle Restart=always RestartSec=3 Environment="OLLAMA_HOST=0.0.0.0:11434" Environment="OLLAMA_MODELS=/home/charle/.ollama/models" Environment="OLLAMA_NUM_GPU=1" [Install] WantedBy=default.target EOF systemctl daemon-reload systemctl enable ollama systemctl restart ollama sleep 2 if systemctl is-active --quiet ollama; then log_success "Servicio Ollama activo" else log_error "Error iniciando Ollama" fi } setup_python_venv() { log "LOG" "Creando Python Virtual Environment..." if [ -d "$VENV_DIR" ]; then log_success "VEnv ya existe" return fi python3.11 -m venv "$VENV_DIR" source "$VENV_DIR/bin/activate" pip install --upgrade pip setuptools wheel pip install \ fastapi \ uvicorn \ python-multipart \ aiofiles \ pillow \ python-dotenv \ requests \ langchain \ pydantic log_success "Python VEnv configurado" } create_services() { log "LOG" "Creando systemd services..." # API Service cat > /etc/systemd/system/llm-api.service << EOF [Unit] Description=LLM API Server After=network.target ollama.service [Service] Type=simple User=charle WorkingDirectory=$PROJECT_DIR Environment="PATH=$VENV_DIR/bin" Environment="OLLAMA_URL=http://localhost:11434" ExecStart=$VENV_DIR/bin/python $PROJECT_DIR/backend.py Restart=always RestartSec=10 [Install] WantedBy=multi-user.target EOF systemctl daemon-reload systemctl enable llm-api log_success "Systemd services creados" } download_models() { log "LOG" "Descargando modelos (esto puede tardar 20-30 minutos)..." log_warning "Esto es OPCIONAL. Presiona Ctrl+C para cancelar." sleep 5 source "$VENV_DIR/bin/activate" # Esperar a que Ollama esté listo for i in {1..30}; do if curl -s http://localhost:11434/api/tags > /dev/null; then log_success "Ollama está listo" break fi log "LOG" "Esperando Ollama... ($i/30)" sleep 2 done log "LOG" "Descargando Phi..." ollama pull phi:latest log "LOG" "Descargando DeepSeek Coder 33B..." ollama pull deepseek-coder:33b log_success "Modelos descargados" } setup_directories() { log "LOG" "Creando directorios..." mkdir -p "$WORKSPACE_DIR" mkdir -p "$PROJECT_DIR/uploads" mkdir -p "$PROJECT_DIR/logs" # Cambiar permisos chown -R charle:charle "$PROJECT_DIR" chmod -R 755 "$PROJECT_DIR" log_success "Directorios creados" } generate_config() { log "LOG" "Generando archivos de configuración..." # .env file if [ ! -f "$PROJECT_DIR/.env" ]; then cat > "$PROJECT_DIR/.env" << EOF # LLM Server Configuration LLM_SERVER_URL=http://localhost:8000 LLM_FAST_MODEL=phi LLM_POWER_MODEL=deepseek-coder:33b OLLAMA_URL=http://localhost:11434 WORKSPACE_DIR=$WORKSPACE_DIR # Server HOST=0.0.0.0 API_PORT=8000 WEB_PORT=5000 # Logging LOG_LEVEL=INFO EOF log_success ".env creado" fi # Config JSON if [ ! -f "$PROJECT_DIR/config/server.json" ]; then mkdir -p "$PROJECT_DIR/config" cat > "$PROJECT_DIR/config/server.json" << 'EOF' { "server": { "host": "0.0.0.0", "api_port": 8000, "web_port": 5000, "workers": 4 }, "models": { "fast": "phi", "power": "deepseek-coder:33b" }, "gpu": { "memory_fraction": 0.9, "max_batch_size": 8 }, "logging": { "level": "INFO", "format": "%(asctime)s - %(name)s - %(levelname)s - %(message)s" } } EOF log_success "config/server.json creado" fi } print_next_steps() { cat << EOF ${BLUE}=========================================${NC} ${GREEN}✨ PRÓXIMOS PASOS:${NC} ${BLUE}=========================================${NC} 1. ${YELLOW}Verificar instalación:${NC} sudo systemctl status ollama sudo systemctl status llm-api 2. ${YELLOW}Ver logs:${NC} sudo journalctl -u ollama -f sudo journalctl -u llm-api -f 3. ${YELLOW}Iniciar servicios:${NC} sudo systemctl start ollama sudo systemctl start llm-api 4. ${YELLOW}Acceder a la API:${NC} curl http://localhost:8000/health 5. ${YELLOW}Descargar modelos (si no se descargaron):${NC} source $VENV_DIR/bin/activate ollama pull phi:latest ollama pull deepseek-coder:33b 6. ${YELLOW}Ver estado en tiempo real:${NC} watch -n 1 nvidia-smi ${BLUE}=========================================${NC} ${GREEN}📁 Archivos importantes:${NC} ${BLUE}=========================================${NC} Config: $PROJECT_DIR/config/server.json .env: $PROJECT_DIR/.env Logs: $PROJECT_DIR/logs/ Workspace: $WORKSPACE_DIR/ ${BLUE}=========================================${NC} EOF } # Ejecutar main main "$@"