From b99ebc7769d2e0c8e1f6bf106ba1021eda891736 Mon Sep 17 00:00:00 2001 From: carlostellocba Date: Sun, 20 Sep 2026 23:01:12 -0300 Subject: [PATCH 1/3] Add setup_llm_server.sh --- setup_llm_server.sh | 0 1 file changed, 0 insertions(+), 0 deletions(-) create mode 100644 setup_llm_server.sh diff --git a/setup_llm_server.sh b/setup_llm_server.sh new file mode 100644 index 0000000..e69de29 From f695480738ddd070a0ebec69865f8b731f9fc169 Mon Sep 17 00:00:00 2001 From: carlostellocba Date: Sun, 20 Sep 2026 23:01:29 -0300 Subject: [PATCH 2/3] Update setup_llm_server.sh --- setup_llm_server.sh | 142 ++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 142 insertions(+) diff --git a/setup_llm_server.sh b/setup_llm_server.sh index e69de29..9b15137 100644 --- a/setup_llm_server.sh +++ b/setup_llm_server.sh @@ -0,0 +1,142 @@ +#!/usr/bin/env bash +set -e + +# Detectar usuario normal si se ejecuta con sudo +TARGET_USER="${SUDO_USER:-$USER}" +TARGET_HOME=$(getent passwd "$TARGET_USER" | cut -d: -f6) + +echo "=== [1/8] Actualizando paquetes base del sistema ===" +export DEBIAN_FRONTEND=noninteractive +apt-get update && apt-get upgrade -y +apt-get install -y curl wget git build-essential ufw jq python3-pip python3-venv smartmontools + +echo "=== [2/8] Configurando Swap de 8 GB ===" +if [ ! -f /swapfile ]; then + fallocate -l 8G /swapfile || dd if=/dev/zero of=/swapfile bs=1M count=8192 + chmod 600 /swapfile + mkswap /swapfile + swapon /swapfile + if ! grep -q '/swapfile' /etc/fstab; then + echo '/swapfile none swap sw 0 0' >> /etc/fstab + fi + sysctl vm.swappiness=10 + echo 'vm.swappiness=10' > /etc/sysctl.d/99-swappiness.conf +fi + +echo "=== [3/8] Instalando Drivers NVIDIA y CUDA ===" +if ! command -v nvidia-smi &> /dev/null; then + apt-get install -y ubuntu-drivers-common + ubuntu-drivers install --gpgpu || apt-get install -y nvidia-driver-535-server +fi + +echo "=== [4/8] Configurando servicio systemd para Power Limit (300W) ===" +cat << 'EOF' > /etc/systemd/system/nvidia-power-limit.service +[Unit] +Description=Set NVIDIA GPU Power Limit to 300W +After=systemd-modules-load.service + +[Service] +Type=oneshot +ExecStart=/usr/bin/nvidia-smi -pm 1 +ExecStart=/usr/bin/nvidia-smi -pl 300 +RemainAfterExit=yes + +[Install] +WantedBy=multi-user.target +EOF + +systemctl daemon-reload +systemctl enable nvidia-power-limit.service +# Intentar aplicar de inmediato si el driver ya está cargado +nvidia-smi -pm 1 2>/dev/null || true +nvidia-smi -pl 300 2>/dev/null || true + +echo "=== [5/8] Instalando Ollama y configurando servicio ===" +if ! command -v ollama &> /dev/null; then + curl -fsSL https://ollama.com/install.sh | sh +fi + +# Configurar variables de Ollama para acceso de red y precarga +mkdir -p /etc/systemd/system/ollama.service.d +cat << 'EOF' > /etc/systemd/system/ollama.service.d/override.conf +[Service] +Environment="OLLAMA_HOST=0.0.0.0:11434" +Environment="OLLAMA_KEEP_ALIVE=-1" +Environment="OLLAMA_NUM_PARALLEL=1" +EOF + +systemctl daemon-reload +systemctl restart ollama +systemctl enable ollama + +echo "=== [6/8] Configurando LiteLLM Proxy en entorno virtual dedicado ===" +rm -rf /opt/litellm-env +python3 -m venv /opt/litellm-env +/opt/litellm-env/bin/pip install --upgrade pip +/opt/litellm-env/bin/pip install 'litellm[proxy]' nvitop + +# Crear configuración de LiteLLM para Tool Calling y VS Code +cat << 'EOF' > "$TARGET_HOME/litellm_config.yaml" +model_list: + - model_name: qwen2.5-coder:14b + litellm_params: + model: ollama/qwen2.5-coder:14b + api_base: http://127.0.0.1:11434 + max_tokens: 32768 + +litellm_settings: + drop_params: true + set_verbose: false +EOF +chown "$TARGET_USER:$TARGET_USER" "$TARGET_HOME/litellm_config.yaml" + +# Crear servicio systemd para LiteLLM +cat << EOF > /etc/systemd/system/litellm.service +[Unit] +Description=LiteLLM Proxy Service for VS Code Agent +After=network.target ollama.service +Wants=ollama.service + +[Service] +Type=simple +User=root +WorkingDirectory=$TARGET_HOME +ExecStart=/opt/litellm-env/bin/litellm --config $TARGET_HOME/litellm_config.yaml --port 8000 --host 0.0.0.0 +Restart=always +RestartSec=5 +Environment=PYTHONUNBUFFERED=1 + +[Install] +WantedBy=multi-user.target +EOF + +systemctl daemon-reload +systemctl enable litellm.service + +echo "=== [7/8] Configurando Firewall (UFW) ===" +ufw allow 22/tcp comment 'SSH' +ufw allow 8000/tcp comment 'LiteLLM Proxy' +ufw --force enable + +echo "=== [8/8] Descargando modelo qwen2.5-coder:14b ===" +# Esperar que Ollama responda antes del pull +until curl -s http://127.0.0.1:11434/api/tags > /dev/null; do + echo "Esperando a que Ollama levante..." + sleep 2 +done + +ollama pull qwen2.5-coder:14b + +# Iniciar LiteLLM ahora que el modelo existe +systemctl start litellm.service + +echo "" +echo "==========================================================" +echo " ¡INSTALACIÓN COMPLETADA EXITOSAMENTE!" +echo "==========================================================" +echo "IP del Servidor: $(hostname -I | awk '{print $1}')" +echo "Endpoint LiteLLM: http://$(hostname -I | awk '{print $1}'):8000" +echo "" +echo "Nota: Si los drivers NVIDIA eran nuevos, se recomienda un reinicio:" +echo "sudo reboot" +echo "==========================================================" \ No newline at end of file From f6ac45d73f1704c2ba1af57443e05c16135050cc Mon Sep 17 00:00:00 2001 From: carlostellocba Date: Sun, 20 Sep 2026 23:08:29 -0300 Subject: [PATCH 3/3] Add README.md --- README.md | 5 +++++ 1 file changed, 5 insertions(+) create mode 100644 README.md diff --git a/README.md b/README.md new file mode 100644 index 0000000..cb2cbc2 --- /dev/null +++ b/README.md @@ -0,0 +1,5 @@ +nano setup_llm_server.sh + +chmod +x setup_llm_server.sh +sudo ./setup_llm_server.sh +