feat(sync): add academic data reset tool, configurable sheets URL and fix sync timeout

This commit is contained in:
2026-09-19 11:13:24 -03:00
parent 42b31ec44e
commit 5087e22988
10 changed files with 835 additions and 101 deletions
+44 -3
View File
@@ -2,6 +2,7 @@ import urllib.request
import csv
import io
import re
import os
from datetime import datetime, date, time, timedelta
from app import db
from app.models.classroom import Classroom
@@ -12,6 +13,34 @@ from app.models.user import User
BASE_CSV_URL = 'https://docs.google.com/spreadsheets/d/e/2PACX-1vSc_T_BQjbn3uPelioCgx52UM5Py-qNhJN0TYPd1kmsN5jdb3Q8rAaIvNMF_2ZTzQt6bH--yIWQKrKR/pub?single=true&output=csv'
def normalize_sheets_url(url):
"""
Normaliza enlaces de Google Sheets (compartidos, de edición o publicados)
para obtener la URL base que acepta parámetro gid=<gid>&output=csv o format=csv.
"""
if not url:
return BASE_CSV_URL
url = url.strip()
# 1. Enlace publicado web (/d/e/2PACX-.../pub...)
if '/d/e/' in url:
base_match = re.match(r'^(https://docs\.google\.com/spreadsheets/d/e/[^/?#]+)/pub', url)
if base_match:
return f"{base_match.group(1)}/pub?single=true&output=csv"
clean = url.split('&gid=')[0].split('?gid=')[0]
sep = '&' if '?' in clean else '?'
if 'output=csv' not in clean:
clean = f"{clean}{sep}single=true&output=csv"
return clean
# 2. Enlace de edición o visualización (/d/<SPREADSHEET_ID>/edit...)
sheet_id_match = re.search(r'/spreadsheets/d/([a-zA-Z0-9-_]+)', url)
if sheet_id_match:
sheet_id = sheet_id_match.group(1)
return f"https://docs.google.com/spreadsheets/d/{sheet_id}/export?format=csv"
return url
SHEETS_CONFIG = [
{ 'name': 'Lunes', 'gid': '1915752353', 'day_idx': 0 },
{ 'name': 'Martes', 'gid': '466778263', 'day_idx': 1 },
@@ -30,16 +59,28 @@ class GoogleSheetsImporter:
"""
def __init__(self, base_url=None):
self.base_url = base_url or BASE_CSV_URL
if base_url:
self.base_url = normalize_sheets_url(base_url)
else:
configured_url = None
try:
from app.models.setting import SystemSetting
configured_url = SystemSetting.get_value('google_sheets_url')
except Exception:
pass
if not configured_url:
configured_url = os.getenv('GOOGLE_SHEETS_URL')
self.base_url = normalize_sheets_url(configured_url) if configured_url else BASE_CSV_URL
def fetch_sheet_csv(self, gid):
"""Fetch CSV string for a specific sheet gid"""
url = f"{self.base_url}&gid={gid}"
separator = '&' if '?' in self.base_url else '?'
url = f"{self.base_url}{separator}gid={gid}"
req = urllib.request.Request(
url,
headers={'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) EduSpace/2.0'}
)
with urllib.request.urlopen(req, timeout=15) as resp:
with urllib.request.urlopen(req, timeout=25) as resp:
return resp.read().decode('utf-8', errors='replace')
def parse_sheet_rows(self, csv_content, sheet_name):