fix: normalizar sync de google sheets para todos los dias, reparar 500 en ciclos lectivos, selector de ciclo lectivo en comisiones y opciones --keep-db --keep-users en install.sh
This commit is contained in:
@@ -13,28 +13,35 @@ from app.models.user import User
|
||||
|
||||
BASE_CSV_URL = 'https://docs.google.com/spreadsheets/d/e/2PACX-1vSc_T_BQjbn3uPelioCgx52UM5Py-qNhJN0TYPd1kmsN5jdb3Q8rAaIvNMF_2ZTzQt6bH--yIWQKrKR/pub?single=true&output=csv'
|
||||
|
||||
def map_day_to_idx(day_str):
|
||||
"""Mapea el nombre del día (o prefijo) a su índice numérico 0..6 (Lunes=0)"""
|
||||
d = (day_str or '').lower()
|
||||
if 'lun' in d: return 0
|
||||
if 'mar' in d: return 1
|
||||
if 'mie' in d or 'mié' in d: return 2
|
||||
if 'jue' in d: return 3
|
||||
if 'vie' in d: return 4
|
||||
if 'sab' in d or 'sáb' in d: return 5
|
||||
if 'dom' in d: return 6
|
||||
return 0
|
||||
|
||||
def normalize_sheets_url(url):
|
||||
"""
|
||||
Normaliza enlaces de Google Sheets (compartidos, de edición o publicados)
|
||||
Normaliza enlaces de Google Sheets (compartidos, de edición, con /u/1/, #gid=... o /pubhtml)
|
||||
para obtener la URL base que acepta parámetro gid=<gid>&output=csv o format=csv.
|
||||
"""
|
||||
if not url:
|
||||
return BASE_CSV_URL
|
||||
url = url.strip()
|
||||
|
||||
# 1. Enlace publicado web (/d/e/2PACX-.../pub...)
|
||||
if '/d/e/' in url:
|
||||
base_match = re.match(r'^(https://docs\.google\.com/spreadsheets/d/e/[^/?#]+)/pub', url)
|
||||
if base_match:
|
||||
return f"{base_match.group(1)}/pub?single=true&output=csv"
|
||||
clean = url.split('&gid=')[0].split('?gid=')[0]
|
||||
sep = '&' if '?' in clean else '?'
|
||||
if 'output=csv' not in clean:
|
||||
clean = f"{clean}{sep}single=true&output=csv"
|
||||
return clean
|
||||
# 1. Enlace publicado web (/d/e/2PACX-... con o sin /u/X/ y con pub o pubhtml)
|
||||
pub_match = re.search(r'/spreadsheets/(?:u/\d+/)?d/e/([a-zA-Z0-9-_]+)', url)
|
||||
if pub_match:
|
||||
doc_id = pub_match.group(1)
|
||||
return f"https://docs.google.com/spreadsheets/d/e/{doc_id}/pub?single=true&output=csv"
|
||||
|
||||
# 2. Enlace de edición o visualización (/d/<SPREADSHEET_ID>/edit...)
|
||||
sheet_id_match = re.search(r'/spreadsheets/d/([a-zA-Z0-9-_]+)', url)
|
||||
# 2. Enlace de edición o exportación (/d/<SPREADSHEET_ID>/...)
|
||||
sheet_id_match = re.search(r'/spreadsheets/(?:u/\d+/)?d/([a-zA-Z0-9-_]+)', url)
|
||||
if sheet_id_match:
|
||||
sheet_id = sheet_id_match.group(1)
|
||||
return f"https://docs.google.com/spreadsheets/d/{sheet_id}/export?format=csv"
|
||||
@@ -72,6 +79,40 @@ class GoogleSheetsImporter:
|
||||
configured_url = os.getenv('GOOGLE_SHEETS_URL')
|
||||
self.base_url = normalize_sheets_url(configured_url) if configured_url else BASE_CSV_URL
|
||||
|
||||
def get_sheets_config(self):
|
||||
"""
|
||||
Descubre las pestañas y GIDs dinámicamente desde /pubhtml si está disponible,
|
||||
o utiliza el fallback SHEETS_CONFIG preconfigurado.
|
||||
"""
|
||||
pub_match = re.search(r'/spreadsheets/d/e/([a-zA-Z0-9-_]+)', self.base_url)
|
||||
if pub_match:
|
||||
doc_id = pub_match.group(1)
|
||||
pubhtml_url = f"https://docs.google.com/spreadsheets/d/e/{doc_id}/pubhtml"
|
||||
try:
|
||||
req = urllib.request.Request(
|
||||
pubhtml_url,
|
||||
headers={'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) EduSpace/2.0'}
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=12) as resp:
|
||||
html_content = resp.read().decode('utf-8', errors='replace')
|
||||
# Extraer pares name y gid de Google Sheets pubhtml
|
||||
found = re.findall(r'name:\s*["\']([^"\']+)["\'],\s*gid:\s*["\'](\d+)["\']', html_content)
|
||||
if found:
|
||||
discovered = []
|
||||
for name, gid in found:
|
||||
clean_name = re.sub(r'[^a-zA-ZáéíóúÁÉÍÓÚñÑ]', '', name).capitalize()
|
||||
d_idx = map_day_to_idx(clean_name)
|
||||
discovered.append({
|
||||
'name': clean_name if clean_name else name,
|
||||
'gid': gid,
|
||||
'day_idx': d_idx
|
||||
})
|
||||
if discovered:
|
||||
return discovered
|
||||
except Exception:
|
||||
pass
|
||||
return SHEETS_CONFIG
|
||||
|
||||
def fetch_sheet_csv(self, gid):
|
||||
"""Fetch CSV string for a specific sheet gid"""
|
||||
separator = '&' if '?' in self.base_url else '?'
|
||||
@@ -110,7 +151,10 @@ class GoogleSheetsImporter:
|
||||
current_shift = 'Mañana'
|
||||
for row in rows[header_idx + 1:]:
|
||||
if len(row) >= col_offset + 5:
|
||||
raw_cell = row[col_offset].strip().replace('\n', ' - ')
|
||||
def _clean(val):
|
||||
return val.replace('\ufb01', 'fi').replace('\ufb02', 'fl').strip()
|
||||
|
||||
raw_cell = _clean(row[col_offset]).replace('\n', ' - ')
|
||||
row_full_text = ' '.join(row).upper()
|
||||
|
||||
# Detect shift delimiters
|
||||
@@ -128,10 +172,10 @@ class GoogleSheetsImporter:
|
||||
continue
|
||||
|
||||
codigo = raw_cell
|
||||
asignatura = row[col_offset + 1].strip().replace('\n', ' ')
|
||||
aula = row[col_offset + 2].strip().replace('\n', ' ')
|
||||
carrera = row[col_offset + 3].strip().replace('\n', ' ')
|
||||
horario = row[col_offset + 4].strip().replace('\n', ' ')
|
||||
asignatura = _clean(row[col_offset + 1]).replace('\n', ' ')
|
||||
aula = _clean(row[col_offset + 2]).replace('\n', ' ')
|
||||
carrera = _clean(row[col_offset + 3]).replace('\n', ' ')
|
||||
horario = _clean(row[col_offset + 4]).replace('\n', ' ')
|
||||
|
||||
# Check valid data row
|
||||
if codigo and asignatura and not 'codigo' in codigo.lower():
|
||||
@@ -250,7 +294,8 @@ class GoogleSheetsImporter:
|
||||
}
|
||||
|
||||
all_raw_records = []
|
||||
for sheet_cfg in SHEETS_CONFIG:
|
||||
sheets_to_process = self.get_sheets_config()
|
||||
for sheet_cfg in sheets_to_process:
|
||||
try:
|
||||
csv_data = self.fetch_sheet_csv(sheet_cfg['gid'])
|
||||
rows = self.parse_sheet_rows(csv_data, sheet_cfg['name'])
|
||||
|
||||
Reference in New Issue
Block a user