+
+
+ """
+
+
+@app.get("/api", tags=["root"])
+def api_info():
"""Endpoint raiz com informações da API."""
return {
"message": "API de Cursos da UAlg",
diff --git a/src/scrape_ualg.py b/src/scrape_ualg.py
index 23f1e24..2e290db 100644
--- a/src/scrape_ualg.py
+++ b/src/scrape_ualg.py
@@ -15,6 +15,7 @@
import os
import time
import logging
+import re
from urllib.parse import urljoin, urlparse
from typing import Optional, List, Dict
from .config import Config
@@ -222,44 +223,144 @@ def parse_course_page(self, soup: BeautifulSoup, url: str) -> Dict:
data["title"] = h1.get_text(strip=True) if h1 else "Título não encontrado"
# Descrição
- desc = soup.select_one(".field--name-body, .text, .course-description")
+ desc = soup.select_one(".field--name-body, .text, .course-description, .description")
data["description"] = desc.get_text(separator="\n", strip=True) if desc else None
# Código do curso
- code_elem = soup.select_one(".field--name-field-code, .course-code")
+ code_elem = soup.select_one(".field--name-field-code, .course-code, .code")
data["code"] = code_elem.get_text(strip=True) if code_elem else None
# Nível (Licenciatura, Mestrado, etc.)
- level_elem = soup.select_one(".field--name-field-level, .course-level")
+ level_elem = soup.select_one(".field--name-field-level, .course-level, .level, .tipo-curso")
data["level"] = level_elem.get_text(strip=True) if level_elem else None
# Escola/Faculdade
- school_elem = soup.select_one(".field--name-field-school, .course-school")
+ school_elem = soup.select_one(".field--name-field-school, .course-school, .school, .escola")
data["school"] = school_elem.get_text(strip=True) if school_elem else None
# Idioma
- lang_elem = soup.select_one(".field--name-field-language, .course-language")
+ lang_elem = soup.select_one(".field--name-field-language, .course-language, .language, .idioma")
data["language"] = lang_elem.get_text(strip=True) if lang_elem else None
# Regime
- regime_elem = soup.select_one(".field--name-field-regime, .course-regime")
+ regime_elem = soup.select_one(".field--name-field-regime, .course-regime, .regime")
data["regime"] = regime_elem.get_text(strip=True) if regime_elem else None
# Modalidade
- modality_elem = soup.select_one(".field--name-field-modality, .course-modality")
+ modality_elem = soup.select_one(".field--name-field-modality, .course-modality, .modality, .modalidade")
data["modality"] = modality_elem.get_text(strip=True) if modality_elem else None
+ # Áreas de conhecimento
+ areas = []
+ area_selectors = [
+ ".field--name-field-area, .course-area, .area, .area-conhecimento",
+ ".field--name-field-areas",
+ ".areas-conhecimento",
+ ]
+ for selector in area_selectors:
+ area_elems = soup.select(selector)
+ for elem in area_elems:
+ area_text = elem.get_text(strip=True)
+ if area_text and area_text not in areas:
+ areas.append(area_text)
+ data["areas"] = areas
+
+ # Módulos/Unidades Curriculares (UCs)
+ modules = []
+
+ # Procurar por tabelas de plano curricular
+ tables = soup.select("table.plano-curricular, table.curriculum, table.modules, table")
+ for table in tables:
+ rows = table.find_all("tr")
+ for row in rows[1:]: # Skip header
+ cells = row.find_all(["td", "th"])
+ if len(cells) >= 2:
+ module = {
+ "code": None,
+ "title": None,
+ "ects": None,
+ "year": None,
+ "semester": None,
+ }
+
+ # Extrair informações das células
+ for i, cell in enumerate(cells):
+ cell_text = cell.get_text(strip=True)
+ if not cell_text:
+ continue
+
+ # Heurísticas para identificar o tipo de dado
+ if i == 0 or "código" in cell_text.lower() or "code" in cell_text.lower():
+ # Possível código
+ if len(cell_text) < 20 and not module["code"]:
+ module["code"] = cell_text
+ elif i == 1 or "disciplina" in cell_text.lower() or "uc" in cell_text.lower():
+ # Possível título
+ if len(cell_text) > 3 and not module["title"]:
+ module["title"] = cell_text
+
+ # ECTS
+ if "ects" in cell_text.lower():
+ try:
+ import re
+ ects_match = re.search(r'(\d+(?:\.\d+)?)', cell_text)
+ if ects_match:
+ module["ects"] = float(ects_match.group(1))
+ except:
+ pass
+
+ # Ano
+ if "ano" in cell_text.lower() or "year" in cell_text.lower():
+ try:
+ import re
+ year_match = re.search(r'(\d+)', cell_text)
+ if year_match:
+ module["year"] = int(year_match.group(1))
+ except:
+ pass
+
+ # Semestre
+ if "semestre" in cell_text.lower() or "semester" in cell_text.lower():
+ try:
+ import re
+ sem_match = re.search(r'(\d+)', cell_text)
+ if sem_match:
+ module["semester"] = int(sem_match.group(1))
+ except:
+ pass
+
+ # Adicionar módulo se tiver pelo menos título
+ if module["title"]:
+ modules.append(module)
+
+ # Também procurar por listas de UCs
+ uc_lists = soup.select("ul.ucs, ul.modules, .curriculum-list")
+ for ul in uc_lists:
+ items = ul.find_all("li")
+ for item in items:
+ text = item.get_text(strip=True)
+ if text and len(text) > 3:
+ modules.append({
+ "code": None,
+ "title": text,
+ "ects": None,
+ "year": None,
+ "semester": None,
+ })
+
+ data["modules"] = modules
+
# Documentos (PDFs)
documents = []
for link in soup.find_all("a", href=True):
href = urljoin(url, link["href"])
- if href.lower().endswith(".pdf"):
+ if href.lower().endswith((".pdf", ".doc", ".docx")):
title = link.get_text(strip=True) or "Documento"
documents.append((title, href))
data["documents"] = documents
- logger.info(f"Curso parseado: {data['title']}")
+ logger.info(f"Curso parseado: {data['title']} (áreas: {len(areas)}, módulos: {len(modules)}, docs: {len(documents)})")
return data
def get_course_links(self, soup: BeautifulSoup) -> List[str]:
@@ -339,6 +440,62 @@ def scrape_all_courses(self, limit: Optional[int] = None) -> None:
course_id = self.save_course(conn, course_data)
if course_id:
+ # Salvar áreas de conhecimento
+ for area_name in course_data.get("areas", []):
+ area_id = self.upsert_area(conn, area_name)
+ if area_id:
+ # Link curso-área
+ cur = conn.cursor()
+ cur.execute(
+ "INSERT OR IGNORE INTO course_area(course_id, area_id) VALUES (?, ?)",
+ (course_id, area_id)
+ )
+ conn.commit()
+
+ # Salvar módulos/UCs
+ for module_data in course_data.get("modules", []):
+ if not module_data.get("title"):
+ continue
+
+ cur = conn.cursor()
+ # Inserir ou recuperar módulo
+ cur.execute(
+ """
+ INSERT OR IGNORE INTO modules(code, title, ects, year, semester)
+ VALUES (?, ?, ?, ?, ?)
+ """,
+ (
+ module_data.get("code"),
+ module_data.get("title"),
+ module_data.get("ects"),
+ module_data.get("year"),
+ module_data.get("semester"),
+ )
+ )
+ conn.commit()
+
+ # Recuperar ID do módulo
+ if module_data.get("code"):
+ cur.execute(
+ "SELECT id FROM modules WHERE code=? AND title=?",
+ (module_data.get("code"), module_data.get("title"))
+ )
+ else:
+ cur.execute(
+ "SELECT id FROM modules WHERE title=? AND code IS NULL",
+ (module_data.get("title"),)
+ )
+
+ result = cur.fetchone()
+ if result:
+ module_id = result[0]
+ # Link módulo-curso
+ cur.execute(
+ "INSERT OR IGNORE INTO module_course(module_id, course_id, mandatory) VALUES (?, ?, ?)",
+ (module_id, course_id, True)
+ )
+ conn.commit()
+
# Salvar documentos
for title, doc_url in course_data.get("documents", []):
self.save_document(conn, course_id, title, doc_url)
diff --git a/templates/index.html b/templates/index.html
new file mode 100644
index 0000000..9f11eaf
--- /dev/null
+++ b/templates/index.html
@@ -0,0 +1,482 @@
+
+
+
+
+
+ Catálogo de Cursos UAlg
+
+
+
+
+
+
📚 Catálogo de Cursos UAlg
+
Universidade do Algarve - Oferta Formativa
+
+
+
+
+
-
+
Cursos
+
+
+
-
+
Módulos
+
+
+
-
+
Documentos
+
+
+
+
+
🔍 Filtros
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
📊 Estatísticas
+
+
+
+
+
+
+ Carregando cursos...
+
+
+
+
+
+
+
+
+
diff --git a/tests/test_api.py b/tests/test_api.py
index 7bb5cac..730af93 100644
--- a/tests/test_api.py
+++ b/tests/test_api.py
@@ -108,9 +108,17 @@ class TestAPI:
"""Testes para endpoints da API."""
def test_read_root(self, client):
- """Testa endpoint raiz."""
+ """Testa endpoint raiz (HTML frontend)."""
response = client.get("/")
assert response.status_code == 200
+ assert "text/html" in response.headers.get("content-type", "")
+ # Verificar que contém elementos HTML esperados
+ assert "" in response.text or "
Date: Mon, 27 Oct 2025 10:27:44 +0000
Subject: [PATCH 3/4] Add demo data script, update docs, and fix linting issues
Co-authored-by: marcelo-m7 <117441129+marcelo-m7@users.noreply.github.com>
---
Makefile | 6 +-
README.md | 57 +++++++-
scripts/populate_demo_data.py | 253 ++++++++++++++++++++++++++++++++++
src/api.py | 3 +-
src/scrape_ualg.py | 71 +++++-----
5 files changed, 344 insertions(+), 46 deletions(-)
create mode 100644 scripts/populate_demo_data.py
diff --git a/Makefile b/Makefile
index e088d0c..28730d1 100644
--- a/Makefile
+++ b/Makefile
@@ -1,4 +1,4 @@
-.PHONY: help install install-dev test lint format clean run run-scraper run-api init-db
+.PHONY: help install install-dev test lint format clean run run-scraper run-api init-db demo
help:
@echo "Available targets:"
@@ -12,6 +12,7 @@ help:
@echo " run-scraper - Run the scraper"
@echo " run-api - Run the API server"
@echo " init-db - Initialize the database"
+ @echo " demo - Populate database with demo data"
install:
pip install -r requirements.txt
@@ -49,3 +50,6 @@ run-api:
init-db:
python -c "from src.scrape_ualg import UAlgCourseScraper; s = UAlgCourseScraper(); s.init_db()"
+
+demo:
+ python scripts/populate_demo_data.py
diff --git a/README.md b/README.md
index fc4b845..fa5c3cb 100644
--- a/README.md
+++ b/README.md
@@ -7,10 +7,14 @@ Sistema completo de scraping e API REST para cursos da Universidade do Algarve (
- 🚀 **Scraper Completo**: Extrai informações detalhadas de cursos da UAlg
- 💾 **Banco de Dados SQLite**: Armazena dados em schema normalizado
- 🌐 **API REST com FastAPI**: Endpoints para consultar cursos, níveis, escolas e áreas
+- 🎨 **Interface Web Moderna**: Frontend responsivo com visualizações interativas
+- 📊 **Estatísticas e Gráficos**: Dashboard com Chart.js para visualização de dados
- 📄 **Download de Documentos**: Faz download automático de PDFs e documentos
+- 🎓 **Extração de Módulos/UCs**: Captura unidades curriculares com ECTS, ano, semestre
+- 🏷️ **Áreas de Conhecimento**: Organiza cursos por áreas temáticas
- 🔧 **Configurável**: Timeouts, retries, user agents personalizáveis
- 📦 **Modular**: Código organizado e separado em módulos
-- ✅ **Testes Completos**: Cobertura abrangente de testes
+- ✅ **Testes Completos**: Cobertura abrangente de testes (36 testes)
- 🔄 **Retry Automático**: Mecanismo de retry para requisições falhadas
- 📝 **Logging Detalhado**: Logs para debugging e monitoramento
- ⚡ **Rate Limiting**: Respeita limites do servidor com delays entre requisições
@@ -81,6 +85,23 @@ scraper.init_db()
## Uso
+### Quick Start com Dados de Demonstração
+
+Para testar rapidamente o sistema com dados de exemplo:
+
+```bash
+# 1. Inicializar banco de dados
+make init-db
+
+# 2. Popular com dados de demonstração
+make demo
+
+# 3. Iniciar servidor API
+make run-api
+
+# 4. Acessar interface web em http://localhost:8000/
+```
+
### Scraper Básico (Original)
```python
@@ -138,10 +159,21 @@ uvicorn src.api:app --reload --host 0.0.0.0 --port 8000
A API estará disponível em `http://localhost:8000`
+#### Interface Web
+
+Acesse `http://localhost:8000/` para visualizar a interface web moderna com:
+- 📊 Dashboard com estatísticas gerais
+- 🔍 Filtros interativos por nível, escola e área
+- 📈 Gráficos de distribuição de cursos
+- 🎴 Cards de cursos com informações detalhadas
+- 📱 Design responsivo
+
#### Documentação Interativa da API
+- Interface Web: `http://localhost:8000/`
- Swagger UI: `http://localhost:8000/docs`
- ReDoc: `http://localhost:8000/redoc`
+- API Info (JSON): `http://localhost:8000/api`
#### Exemplos de Uso da API
@@ -165,6 +197,11 @@ curl "http://localhost:8000/courses?level=Licenciatura"
curl "http://localhost:8000/courses?school=Faculdade%20de%20Ciências"
```
+**Filtrar cursos por área:**
+```bash
+curl "http://localhost:8000/courses?area=Tecnologias%20da%20Informação"
+```
+
**Obter estatísticas:**
```bash
curl http://localhost:8000/stats
@@ -241,6 +278,8 @@ Scrape-UAlg-Courses/
│ ├── scraper.py # Scraper básico (original)
│ ├── scrape_ualg.py # Scraper completo com BD
│ └── api.py # API REST FastAPI
+├── templates/
+│ └── index.html # Interface web moderna
├── tests/
│ ├── __init__.py
│ ├── test_scraper.py # Testes do scraper básico
@@ -318,9 +357,10 @@ O scraper pode ser configurado usando a classe `Config` ou variáveis de ambient
| Método | Endpoint | Descrição |
|--------|----------|-----------|
-| GET | `/` | Informações da API |
+| GET | `/` | Interface web moderna (HTML) |
+| GET | `/api` | Informações da API (JSON) |
| GET | `/courses` | Listar cursos (com filtros opcionais) |
-| GET | `/courses/{id}` | Obter detalhes de um curso |
+| GET | `/courses/{id}` | Obter detalhes de um curso com módulos, áreas e documentos |
| GET | `/levels` | Listar níveis de curso |
| GET | `/schools` | Listar escolas/faculdades |
| GET | `/areas` | Listar áreas de conhecimento |
@@ -331,7 +371,7 @@ O scraper pode ser configurado usando a classe `Config` ou variáveis de ambient
- `level`: Filtrar por nível (ex: "Licenciatura", "Mestrado")
- `school`: Filtrar por escola
- `area`: Filtrar por área de conhecimento
-- `limit`: Número máximo de resultados (padrão: 100)
+- `limit`: Número máximo de resultados (padrão: 100, máx: 500)
- `offset`: Offset para paginação (padrão: 0)
## Boas Práticas e Considerações
@@ -374,7 +414,7 @@ Se encontrar problemas ou tiver dúvidas:
## Agradecimentos
-- Construído com [Requests](https://requests.readthedocs.io/), [Beautiful Soup](https://www.crummy.com/software/BeautifulSoup/) e [FastAPI](https://fastapi.tiangolo.com/)
+- Construído com [Requests](https://requests.readthedocs.io/), [Beautiful Soup](https://www.crummy.com/software/BeautifulSoup/), [FastAPI](https://fastapi.tiangolo.com/) e [Chart.js](https://www.chartjs.org/)
- Inspirado pela necessidade de fácil acesso às informações de cursos da UAlg
## Roadmap
@@ -383,7 +423,12 @@ Se encontrar problemas ou tiver dúvidas:
- [x] Banco de dados SQLite normalizado
- [x] API REST completa
- [x] Download automático de documentos
-- [x] Testes completos
+- [x] Testes completos (36 testes)
+- [x] Extração de módulos/UCs com ECTS, ano, semestre
+- [x] Extração e organização por áreas de conhecimento
+- [x] Interface web moderna e responsiva
+- [x] Dashboard com estatísticas e gráficos
+- [x] Filtros interativos por nível, escola e área
- [ ] Interface de linha de comando (CLI)
- [ ] Exportação de dados (JSON, CSV, Excel)
- [ ] Suporte para agendamento automático
diff --git a/scripts/populate_demo_data.py b/scripts/populate_demo_data.py
new file mode 100644
index 0000000..5ab7c9e
--- /dev/null
+++ b/scripts/populate_demo_data.py
@@ -0,0 +1,253 @@
+#!/usr/bin/env python3
+"""
+Demo script para popular o banco de dados com dados de exemplo.
+
+Este script adiciona cursos, módulos e áreas de exemplo no banco de dados
+para demonstrar a funcionalidade do sistema sem precisar fazer scraping real.
+"""
+
+import sqlite3
+import os
+import sys
+
+# Adicionar o diretório src ao path
+sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
+
+from src.scrape_ualg import UAlgCourseScraper
+
+def populate_demo_data():
+ """Popula o banco de dados com dados de exemplo."""
+
+ print("🚀 Iniciando população do banco de dados com dados de exemplo...")
+
+ # Inicializar scraper e banco de dados
+ scraper = UAlgCourseScraper(db_path="ualg_courses.db")
+ scraper.init_db()
+
+ conn = sqlite3.connect("ualg_courses.db")
+ conn.execute("PRAGMA foreign_keys = ON")
+
+ try:
+ # Limpar dados existentes (opcional)
+ print("🧹 Limpando dados anteriores...")
+ conn.execute("DELETE FROM course_documents")
+ conn.execute("DELETE FROM module_course")
+ conn.execute("DELETE FROM course_area")
+ conn.execute("DELETE FROM courses")
+ conn.execute("DELETE FROM modules")
+ conn.execute("DELETE FROM areas")
+ conn.execute("DELETE FROM schools")
+ conn.execute("DELETE FROM levels")
+ conn.commit()
+
+ # Adicionar níveis
+ print("📚 Adicionando níveis...")
+ levels = ["Licenciatura", "Mestrado", "Doutoramento", "Pós-Graduação"]
+ level_ids = {}
+ for level in levels:
+ level_ids[level] = scraper.upsert_level(conn, level)
+
+ # Adicionar escolas
+ print("🏫 Adicionando escolas...")
+ schools = [
+ "Faculdade de Ciências e Tecnologia",
+ "Faculdade de Economia",
+ "Escola Superior de Saúde",
+ "Escola Superior de Gestão, Hotelaria e Turismo"
+ ]
+ school_ids = {}
+ for school in schools:
+ school_ids[school] = scraper.upsert_school(conn, school)
+
+ # Adicionar áreas
+ print("🔬 Adicionando áreas de conhecimento...")
+ areas = [
+ "Tecnologias da Informação",
+ "Engenharia",
+ "Ciências Económicas",
+ "Saúde",
+ "Turismo e Hospitalidade"
+ ]
+ area_ids = {}
+ for area in areas:
+ area_ids[area] = scraper.upsert_area(conn, area)
+
+ # Adicionar cursos de exemplo
+ print("🎓 Adicionando cursos de exemplo...")
+
+ courses = [
+ {
+ "code": "L001",
+ "title": "Engenharia Informática",
+ "description": "Curso de licenciatura em Engenharia Informática, focado no desenvolvimento de software, sistemas de informação e redes de computadores.",
+ "level": "Licenciatura",
+ "school": "Faculdade de Ciências e Tecnologia",
+ "language": "Português",
+ "regime": "Diurno",
+ "modality": "Presencial",
+ "areas": ["Tecnologias da Informação", "Engenharia"],
+ "modules": [
+ {"code": "INF101", "title": "Programação I", "ects": 6.0, "year": 1, "semester": 1},
+ {"code": "INF102", "title": "Matemática Discreta", "ects": 6.0, "year": 1, "semester": 1},
+ {"code": "INF201", "title": "Bases de Dados", "ects": 6.0, "year": 2, "semester": 1},
+ {"code": "INF301", "title": "Engenharia de Software", "ects": 6.0, "year": 3, "semester": 1},
+ ]
+ },
+ {
+ "code": "M001",
+ "title": "Mestrado em Inteligência Artificial",
+ "description": "Mestrado focado em técnicas avançadas de IA, machine learning e processamento de linguagem natural.",
+ "level": "Mestrado",
+ "school": "Faculdade de Ciências e Tecnologia",
+ "language": "Inglês",
+ "regime": "Pós-Laboral",
+ "modality": "Misto",
+ "areas": ["Tecnologias da Informação"],
+ "modules": [
+ {"code": "AI101", "title": "Machine Learning", "ects": 7.5, "year": 1, "semester": 1},
+ {"code": "AI102", "title": "Deep Learning", "ects": 7.5, "year": 1, "semester": 1},
+ {"code": "AI201", "title": "Natural Language Processing", "ects": 7.5, "year": 1, "semester": 2},
+ ]
+ },
+ {
+ "code": "L002",
+ "title": "Economia",
+ "description": "Licenciatura em Economia com forte componente em economia aplicada, finanças e métodos quantitativos.",
+ "level": "Licenciatura",
+ "school": "Faculdade de Economia",
+ "language": "Português",
+ "regime": "Diurno",
+ "modality": "Presencial",
+ "areas": ["Ciências Económicas"],
+ "modules": [
+ {"code": "ECO101", "title": "Microeconomia", "ects": 6.0, "year": 1, "semester": 1},
+ {"code": "ECO102", "title": "Macroeconomia", "ects": 6.0, "year": 1, "semester": 2},
+ {"code": "ECO201", "title": "Econometria", "ects": 6.0, "year": 2, "semester": 1},
+ ]
+ },
+ {
+ "code": "L003",
+ "title": "Enfermagem",
+ "description": "Licenciatura em Enfermagem preparando profissionais para cuidados de saúde de qualidade.",
+ "level": "Licenciatura",
+ "school": "Escola Superior de Saúde",
+ "language": "Português",
+ "regime": "Diurno",
+ "modality": "Presencial",
+ "areas": ["Saúde"],
+ "modules": [
+ {"code": "ENF101", "title": "Anatomia e Fisiologia", "ects": 7.0, "year": 1, "semester": 1},
+ {"code": "ENF102", "title": "Fundamentos de Enfermagem", "ects": 6.0, "year": 1, "semester": 1},
+ {"code": "ENF201", "title": "Enfermagem Médico-Cirúrgica", "ects": 8.0, "year": 2, "semester": 1},
+ ]
+ },
+ {
+ "code": "L004",
+ "title": "Gestão Hoteleira",
+ "description": "Licenciatura em Gestão Hoteleira com foco em operações, marketing e gestão de recursos.",
+ "level": "Licenciatura",
+ "school": "Escola Superior de Gestão, Hotelaria e Turismo",
+ "language": "Português",
+ "regime": "Diurno",
+ "modality": "Presencial",
+ "areas": ["Turismo e Hospitalidade"],
+ "modules": [
+ {"code": "TUR101", "title": "Introdução ao Turismo", "ects": 5.0, "year": 1, "semester": 1},
+ {"code": "TUR102", "title": "Operações Hoteleiras", "ects": 6.0, "year": 1, "semester": 1},
+ {"code": "TUR201", "title": "Marketing Turístico", "ects": 6.0, "year": 2, "semester": 1},
+ ]
+ }
+ ]
+
+ for course in courses:
+ # Criar dados do curso
+ course_data = {
+ "url": f"https://www.ualg.pt/curso/{course['code'].lower()}",
+ "code": course["code"],
+ "title": course["title"],
+ "description": course["description"],
+ "language": course["language"],
+ "regime": course["regime"],
+ "modality": course["modality"],
+ "level_id": level_ids[course["level"]],
+ "school_id": school_ids[course["school"]],
+ }
+
+ # Salvar curso
+ course_id = scraper.save_course(conn, course_data)
+ print(f" ✅ {course['title']}")
+
+ # Adicionar áreas ao curso
+ for area_name in course["areas"]:
+ area_id = area_ids[area_name]
+ conn.execute(
+ "INSERT OR IGNORE INTO course_area(course_id, area_id) VALUES (?, ?)",
+ (course_id, area_id)
+ )
+
+ # Adicionar módulos
+ for module_data in course["modules"]:
+ cur = conn.cursor()
+ cur.execute(
+ """
+ INSERT OR IGNORE INTO modules(code, title, ects, year, semester)
+ VALUES (?, ?, ?, ?, ?)
+ """,
+ (
+ module_data["code"],
+ module_data["title"],
+ module_data["ects"],
+ module_data["year"],
+ module_data["semester"],
+ )
+ )
+ conn.commit()
+
+ # Recuperar ID do módulo
+ cur.execute(
+ "SELECT id FROM modules WHERE code=?",
+ (module_data["code"],)
+ )
+ result = cur.fetchone()
+ if result:
+ module_id = result[0]
+ # Link módulo-curso
+ cur.execute(
+ "INSERT OR IGNORE INTO module_course(module_id, course_id, mandatory) VALUES (?, ?, ?)",
+ (module_id, course_id, True)
+ )
+ conn.commit()
+
+ print("\n✨ Banco de dados populado com sucesso!")
+ print(f"\n📊 Estatísticas:")
+
+ # Mostrar estatísticas
+ cur = conn.cursor()
+ cur.execute("SELECT COUNT(*) FROM courses")
+ print(f" - Cursos: {cur.fetchone()[0]}")
+
+ cur.execute("SELECT COUNT(*) FROM modules")
+ print(f" - Módulos: {cur.fetchone()[0]}")
+
+ cur.execute("SELECT COUNT(*) FROM areas")
+ print(f" - Áreas: {cur.fetchone()[0]}")
+
+ cur.execute("SELECT COUNT(*) FROM schools")
+ print(f" - Escolas: {cur.fetchone()[0]}")
+
+ cur.execute("SELECT COUNT(*) FROM levels")
+ print(f" - Níveis: {cur.fetchone()[0]}")
+
+ print("\n🌐 Inicie o servidor API com: make run-api")
+ print("📱 Acesse a interface web em: http://localhost:8000/")
+
+ except Exception as e:
+ print(f"\n❌ Erro: {e}")
+ import traceback
+ traceback.print_exc()
+ finally:
+ conn.close()
+ scraper.close()
+
+if __name__ == "__main__":
+ populate_demo_data()
diff --git a/src/api.py b/src/api.py
index e99f28a..bb1cf93 100644
--- a/src/api.py
+++ b/src/api.py
@@ -11,7 +11,6 @@
from fastapi import FastAPI, HTTPException, Query
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import HTMLResponse
-from fastapi.staticfiles import StaticFiles
import sqlite3
import os
from typing import List, Optional, Dict, Any
@@ -104,7 +103,7 @@ async def read_root():
"""Serve a interface web principal."""
templates_dir = os.path.join(os.path.dirname(__file__), "..", "templates")
index_path = os.path.join(templates_dir, "index.html")
-
+
if os.path.exists(index_path):
with open(index_path, "r", encoding="utf-8") as f:
return f.read()
diff --git a/src/scrape_ualg.py b/src/scrape_ualg.py
index 2e290db..e5104c0 100644
--- a/src/scrape_ualg.py
+++ b/src/scrape_ualg.py
@@ -267,7 +267,7 @@ def parse_course_page(self, soup: BeautifulSoup, url: str) -> Dict:
# Módulos/Unidades Curriculares (UCs)
modules = []
-
+
# Procurar por tabelas de plano curricular
tables = soup.select("table.plano-curricular, table.curriculum, table.modules, table")
for table in tables:
@@ -282,13 +282,13 @@ def parse_course_page(self, soup: BeautifulSoup, url: str) -> Dict:
"year": None,
"semester": None,
}
-
+
# Extrair informações das células
for i, cell in enumerate(cells):
cell_text = cell.get_text(strip=True)
if not cell_text:
continue
-
+
# Heurísticas para identificar o tipo de dado
if i == 0 or "código" in cell_text.lower() or "code" in cell_text.lower():
# Possível código
@@ -298,41 +298,38 @@ def parse_course_page(self, soup: BeautifulSoup, url: str) -> Dict:
# Possível título
if len(cell_text) > 3 and not module["title"]:
module["title"] = cell_text
-
+
# ECTS
if "ects" in cell_text.lower():
try:
- import re
- ects_match = re.search(r'(\d+(?:\.\d+)?)', cell_text)
+ ects_match = re.search(r"(\d+(?:\.\d+)?)", cell_text)
if ects_match:
module["ects"] = float(ects_match.group(1))
- except:
+ except (ValueError, AttributeError):
pass
-
+
# Ano
if "ano" in cell_text.lower() or "year" in cell_text.lower():
try:
- import re
- year_match = re.search(r'(\d+)', cell_text)
+ year_match = re.search(r"(\d+)", cell_text)
if year_match:
module["year"] = int(year_match.group(1))
- except:
+ except (ValueError, AttributeError):
pass
-
+
# Semestre
if "semestre" in cell_text.lower() or "semester" in cell_text.lower():
try:
- import re
- sem_match = re.search(r'(\d+)', cell_text)
+ sem_match = re.search(r"(\d+)", cell_text)
if sem_match:
module["semester"] = int(sem_match.group(1))
- except:
+ except (ValueError, AttributeError):
pass
-
+
# Adicionar módulo se tiver pelo menos título
if module["title"]:
modules.append(module)
-
+
# Também procurar por listas de UCs
uc_lists = soup.select("ul.ucs, ul.modules, .curriculum-list")
for ul in uc_lists:
@@ -340,14 +337,16 @@ def parse_course_page(self, soup: BeautifulSoup, url: str) -> Dict:
for item in items:
text = item.get_text(strip=True)
if text and len(text) > 3:
- modules.append({
- "code": None,
- "title": text,
- "ects": None,
- "year": None,
- "semester": None,
- })
-
+ modules.append(
+ {
+ "code": None,
+ "title": text,
+ "ects": None,
+ "year": None,
+ "semester": None,
+ }
+ )
+
data["modules"] = modules
# Documentos (PDFs)
@@ -447,16 +446,15 @@ def scrape_all_courses(self, limit: Optional[int] = None) -> None:
# Link curso-área
cur = conn.cursor()
cur.execute(
- "INSERT OR IGNORE INTO course_area(course_id, area_id) VALUES (?, ?)",
- (course_id, area_id)
+ "INSERT OR IGNORE INTO course_area(course_id, area_id) VALUES (?, ?)", (course_id, area_id)
)
conn.commit()
-
+
# Salvar módulos/UCs
for module_data in course_data.get("modules", []):
if not module_data.get("title"):
continue
-
+
cur = conn.cursor()
# Inserir ou recuperar módulo
cur.execute(
@@ -470,32 +468,31 @@ def scrape_all_courses(self, limit: Optional[int] = None) -> None:
module_data.get("ects"),
module_data.get("year"),
module_data.get("semester"),
- )
+ ),
)
conn.commit()
-
+
# Recuperar ID do módulo
if module_data.get("code"):
cur.execute(
"SELECT id FROM modules WHERE code=? AND title=?",
- (module_data.get("code"), module_data.get("title"))
+ (module_data.get("code"), module_data.get("title")),
)
else:
cur.execute(
- "SELECT id FROM modules WHERE title=? AND code IS NULL",
- (module_data.get("title"),)
+ "SELECT id FROM modules WHERE title=? AND code IS NULL", (module_data.get("title"),)
)
-
+
result = cur.fetchone()
if result:
module_id = result[0]
# Link módulo-curso
cur.execute(
"INSERT OR IGNORE INTO module_course(module_id, course_id, mandatory) VALUES (?, ?, ?)",
- (module_id, course_id, True)
+ (module_id, course_id, True),
)
conn.commit()
-
+
# Salvar documentos
for title, doc_url in course_data.get("documents", []):
self.save_document(conn, course_id, title, doc_url)
From d834e2def38d911577c0f535a38a11d86ec89264 Mon Sep 17 00:00:00 2001
From: Copilot <198982749+Copilot@users.noreply.github.com>
Date: Mon, 27 Oct 2025 11:15:13 +0000
Subject: [PATCH 4/4] Add GitHub Copilot environment setup and increase test
coverage to 87% (#4)
* Checkpoint from VS Code for coding agent session
* Add Copilot setup workflow and improve test coverage to 87%
Co-authored-by: marcelo-m7 <117441129+marcelo-m7@users.noreply.github.com>
* Address code review feedback - improve test assertions
Co-authored-by: marcelo-m7 <117441129+marcelo-m7@users.noreply.github.com>
---------
Co-authored-by: marcelo-m7
Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com>
Co-authored-by: marcelo-m7 <117441129+marcelo-m7@users.noreply.github.com>
---
.github/copilot-instructions.md | 41 +++++
.github/workflows/copilot-setup-steps.yml | 51 +++++++
tests/test_api.py | 47 ++++++
tests/test_scrape_ualg.py | 173 ++++++++++++++++++++++
4 files changed, 312 insertions(+)
create mode 100644 .github/copilot-instructions.md
create mode 100644 .github/workflows/copilot-setup-steps.yml
diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md
new file mode 100644
index 0000000..2a739b9
--- /dev/null
+++ b/.github/copilot-instructions.md
@@ -0,0 +1,41 @@
+# Copilot instructions for Scrape-UAlg-Courses
+
+Goal: a complete scraper + REST API for UAlg courses. Source pages are scraped into a normalized SQLite DB; the FastAPI service reads from that DB and serves a simple web UI.
+
+## Architecture and data flow
+- Scraper (basic) in `src/scraper.py` (class `UAlgScraper`): fetches HTML and parses simple "course" blocks. Used mainly in unit tests/examples.
+- Scraper (full) in `src/scrape_ualg.py` (class `UAlgCourseScraper`):
+ - Initializes DB from `schema.sql` via `init_db()`.
+ - Crawls listing pages starting at `START_URL`, extracts course links, parses each course page into fields: title, code, level, school, language, modules, areas, documents.
+ - Persists to SQLite (`ualg_courses.db` by default), also downloads documents to `data/docs/`.
+ - Respects retries (`Config.max_retries`) and rate limiting (sleep ~1.5s between courses).
+- API in `src/api.py` (FastAPI): reads from the same SQLite DB (`DB_PATH = "ualg_courses.db"`) and serves endpoints (`/courses`, `/courses/{id}`, `/levels`, `/schools`, `/areas`, `/stats`, `/api`, `/`). The `/` route serves `templates/index.html`.
+- Frontend in `templates/index.html`: fetches API endpoints and renders dashboards with Chart.js.
+
+## Dev workflows (Windows PowerShell friendly)
+- Install deps: `make install`; dev deps: `make install-dev`.
+- Initialize DB: `make init-db`; load demo data: `make demo` (runs `scripts/populate_demo_data.py`).
+- Run scraper: `make run-scraper` (equivalent to `python -m src.scrape_ualg`).
+- Run API: `make run-api` (uvicorn `src.api:app` on port 8000).
+- Tests: `make test` (pytest + coverage). Lint/format: `make lint`, `make format`.
+
+## Conventions and patterns
+- Config is centralized in `src/config.py` (`Config`): defaults to `https://www.ualg.pt`, timeout=30, max_retries=3, UA set; override with env `UALG_BASE_URL` or via constructor.
+- Logging: both scrapers log progress; keep user-agent and backoff semantics; do not remove the ~1.5s delay in the full scraper.
+- DB access in API: use `get_db_connection()` with `row_factory = sqlite3.Row` and helper `dict_from_row`.
+- Filtering: `/courses` builds SQL dynamically with optional joins on `levels`, `schools`, and `areas`; keep `limit` (1..500) and `offset` semantics.
+- Schema is authoritative: see `schema.sql`. M2M tables: `course_area`, `module_course`; uniqueness and indices are already defined.
+- Paths: code locates `schema.sql` and `templates/index.html` via `..` from `src/`; keep relative paths when adding files.
+
+## When extending
+- API: mirror existing patterns (Pydantic response models at top; small helpers; parameterized SQL; close connections). Add tags and docs to endpoints.
+- Scraper: prefer adding selectors heuristically (see `parse_course_page`), and persist via existing upsert helpers; keep downloads in `data/docs/`.
+- Tests: look at `tests/test_api.py`, `tests/test_scraper.py`, `tests/test_scrape_ualg.py` for expected behavior and monkeypatch patterns (e.g., override `DB_PATH` in API tests).
+
+## Quick examples
+- Add an API filter: extend `list_courses` by appending the join + where + param; preserve `limit/offset` bounds.
+- Add a new field to courses: update `schema.sql` + insert in `save_course` + select in API queries + include in models.
+
+Notes
+- Default DB filename is `ualg_courses.db` in repo root. Populate it via scraper or `scripts/populate_demo_data.py` before using the UI.
+- CI: `.github/workflows/python-app.yml` runs tests/lint on PRs; match the Makefile targets.
diff --git a/.github/workflows/copilot-setup-steps.yml b/.github/workflows/copilot-setup-steps.yml
new file mode 100644
index 0000000..6665e37
--- /dev/null
+++ b/.github/workflows/copilot-setup-steps.yml
@@ -0,0 +1,51 @@
+name: "Copilot Setup Steps"
+
+# Automatically run the setup steps when they are changed to allow for easy validation,
+# and allow manual testing through the repository's "Actions" tab
+on:
+ workflow_dispatch:
+ push:
+ paths:
+ - .github/workflows/copilot-setup-steps.yml
+ pull_request:
+ paths:
+ - .github/workflows/copilot-setup-steps.yml
+
+jobs:
+ # The job MUST be called copilot-setup-steps or it will not be picked up by Copilot.
+ copilot-setup-steps:
+ runs-on: ubuntu-latest
+
+ # Set the permissions to the lowest permissions possible needed for your steps.
+ # Copilot will be given its own token for its operations.
+ permissions:
+ # If you want to clone the repository as part of your setup steps, for example to install dependencies,
+ # you'll need the `contents: read` permission. If you don't clone the repository in your setup steps,
+ # Copilot will do this for you automatically after the steps complete.
+ contents: read
+
+ # You can define any steps you want, and they will run before the agent starts.
+ # If you do not check out your code, Copilot will do this for you.
+ steps:
+ - name: Checkout code
+ uses: actions/checkout@v4
+
+ - name: Set up Python 3.10
+ uses: actions/setup-python@v5
+ with:
+ python-version: "3.10"
+ cache: "pip"
+
+ - name: Install Python dependencies
+ run: |
+ python -m pip install --upgrade pip
+ pip install -r requirements.txt
+ pip install -r requirements-dev.txt
+
+ - name: Initialize database
+ run: |
+ python -c "from src.scrape_ualg import UAlgCourseScraper; s = UAlgCourseScraper(); s.init_db()"
+
+ - name: Verify installation
+ run: |
+ python -c "import requests; import bs4; import fastapi; import pytest; print('All dependencies installed successfully')"
diff --git a/tests/test_api.py b/tests/test_api.py
index 730af93..d50db83 100644
--- a/tests/test_api.py
+++ b/tests/test_api.py
@@ -4,6 +4,7 @@
import pytest
import sqlite3
+import os
from fastapi.testclient import TestClient
from src.api import app
@@ -194,3 +195,49 @@ def test_get_stats(self, client):
assert "total_modules" in data
assert "total_documents" in data
assert data["total_courses"] >= 2
+
+ def test_list_courses_with_school_filter(self, client):
+ """Testa listagem de cursos com filtro de escola."""
+ response = client.get("/courses?school=Escola de Tecnologia")
+ assert response.status_code == 200
+ data = response.json()
+ assert isinstance(data, list)
+ # Deve retornar cursos da escola especificada
+ assert len(data) >= 1
+ # Verificar que todos os cursos têm uma escola associada
+ for course in data:
+ assert course.get("school_id") is not None
+
+ def test_list_courses_with_area_filter(self, client):
+ """Testa listagem de cursos com filtro de área."""
+ response = client.get("/courses?area=Informática")
+ assert response.status_code == 200
+ data = response.json()
+ assert isinstance(data, list)
+ # Deve retornar cursos da área especificada
+ assert len(data) >= 1
+
+ def test_list_courses_with_offset(self, client):
+ """Testa listagem de cursos com paginação (offset)."""
+ response = client.get("/courses?offset=1&limit=1")
+ assert response.status_code == 200
+ data = response.json()
+ assert isinstance(data, list)
+ assert len(data) <= 1
+
+ def test_read_root_fallback(self, client, monkeypatch):
+ """Testa fallback da página inicial quando index.html não existe."""
+ # Mock os.path.exists para retornar False
+ original_exists = os.path.exists
+
+ def mock_exists(path):
+ if "index.html" in path:
+ return False
+ return original_exists(path)
+
+ monkeypatch.setattr(os.path, "exists", mock_exists)
+
+ response = client.get("/")
+ assert response.status_code == 200
+ assert "API de Cursos da UAlg" in response.text
+ assert "/docs" in response.text
diff --git a/tests/test_scrape_ualg.py b/tests/test_scrape_ualg.py
index abbbdce..2ccc533 100644
--- a/tests/test_scrape_ualg.py
+++ b/tests/test_scrape_ualg.py
@@ -209,3 +209,176 @@ def test_close(self, scraper):
"""Testa fechamento da sessão."""
scraper.close()
# Não deve gerar exceção
+
+ @patch("src.scrape_ualg.requests.Session.get")
+ def test_save_document(self, mock_get, scraper, test_db_path):
+ """Testa salvamento de documentos."""
+ scraper.init_db()
+
+ # Mock da resposta HTTP para download
+ mock_response = Mock()
+ mock_response.content = b"PDF content"
+ mock_response.raise_for_status = Mock()
+ mock_get.return_value = mock_response
+
+ conn = sqlite3.connect(test_db_path)
+ conn.execute("PRAGMA foreign_keys = ON")
+
+ # Inserir curso de teste
+ cur = conn.cursor()
+ cur.execute("INSERT INTO levels (name) VALUES ('Teste')")
+ level_id = cur.lastrowid
+ cur.execute("INSERT INTO courses (code, title, level_id) VALUES (?, ?, ?)", ("TEST", "Curso Teste", level_id))
+ course_id = cur.lastrowid
+ conn.commit()
+
+ # Salvar documento
+ scraper.save_document(conn, course_id, "Plano Curricular", "https://test.com/doc.pdf")
+
+ # Verificar se foi salvo
+ cur.execute("SELECT * FROM course_documents WHERE course_id = ?", (course_id,))
+ doc = cur.fetchone()
+
+ assert doc is not None
+ conn.close()
+
+ @patch("src.scrape_ualg.requests.Session.get")
+ def test_save_document_download_error(self, mock_get, scraper, test_db_path):
+ """Testa salvamento de documento com erro no download."""
+ scraper.init_db()
+
+ # Mock erro no download
+ mock_get.side_effect = requests.RequestException("Connection error")
+
+ conn = sqlite3.connect(test_db_path)
+ conn.execute("PRAGMA foreign_keys = ON")
+
+ # Inserir curso de teste
+ cur = conn.cursor()
+ cur.execute("INSERT INTO levels (name) VALUES ('Teste')")
+ level_id = cur.lastrowid
+ cur.execute("INSERT INTO courses (code, title, level_id) VALUES (?, ?, ?)", ("TEST", "Curso Teste", level_id))
+ course_id = cur.lastrowid
+ conn.commit()
+
+ # Tentar salvar documento (deve lidar com erro graciosamente)
+ scraper.save_document(conn, course_id, "Plano Curricular", "https://test.com/doc.pdf")
+
+ # Verificar se o registro foi criado mesmo sem download
+ cur.execute("SELECT * FROM course_documents WHERE course_id = ?", (course_id,))
+ doc = cur.fetchone()
+
+ assert doc is not None
+ conn.close()
+
+ def test_parse_course_page_with_modules(self, scraper):
+ """Testa parsing de página de curso com módulos."""
+ html = """
+
+
+
Curso de Teste
+
Descrição do curso de teste
+
+
+
Código
+
Disciplina
+
ECTS
+
+
+
CS101
+
Programação I
+
6 ECTS
+
+
+
CS102
+
Algoritmos
+
5 ECTS
+
+
+
+
+ """
+
+ soup = BeautifulSoup(html, "html.parser")
+ data = scraper.parse_course_page(soup, "https://test.com/curso")
+
+ assert "modules" in data
+ assert len(data["modules"]) == 2
+ # Verificar que os módulos específicos foram extraídos
+ module_titles = [m["title"] for m in data["modules"]]
+ assert "Programação I" in module_titles
+ assert "Algoritmos" in module_titles
+
+ def test_parse_course_page_with_areas(self, scraper):
+ """Testa parsing de página de curso com áreas."""
+ html = """
+
+
+
Curso de Teste
+
Informática
+
Ciências da Computação
+
+
+ """
+
+ soup = BeautifulSoup(html, "html.parser")
+ data = scraper.parse_course_page(soup, "https://test.com/curso")
+
+ assert "areas" in data
+ assert len(data["areas"]) == 2
+ # Verificar que as áreas específicas foram extraídas
+ assert "Informática" in data["areas"]
+ assert "Ciências da Computação" in data["areas"]
+
+ @patch("src.scrape_ualg.requests.Session.get")
+ def test_scrape_all_courses_with_limit(self, mock_get, scraper, test_db_path):
+ """Testa scraping com limite de cursos."""
+ scraper.init_db()
+
+ # Mock página inicial
+ listing_html = """
+
+
+ Curso 1
+ Curso 2
+ Curso 3
+
+
+ """
+
+ course_html = """
+
+
+
Curso Teste
+
Descrição
+
+
+ """
+
+ mock_response = Mock()
+ mock_response.text = listing_html
+ mock_response.raise_for_status = Mock()
+
+ mock_course_response = Mock()
+ mock_course_response.text = course_html
+ mock_course_response.raise_for_status = Mock()
+
+ # Primeira chamada retorna listagem, demais retornam página de curso
+ mock_get.side_effect = [mock_response] + [mock_course_response] * 3
+
+ # Executar scraping com limite
+ scraper.scrape_all_courses(limit=2)
+
+ # Verificar que apenas 2 cursos foram processados (além da página inicial)
+ assert mock_get.call_count == 3 # 1 listagem + 2 cursos
+
+ @patch("src.scrape_ualg.requests.Session.get")
+ def test_scrape_all_courses_error_handling(self, mock_get, scraper, test_db_path):
+ """Testa tratamento de erros durante scraping."""
+ scraper.init_db()
+
+ # Mock erro na página inicial (simular exceção de rede)
+ mock_get.side_effect = requests.RequestException("Connection error")
+
+ # Não deve lançar exceção
+ scraper.scrape_all_courses(limit=1)