diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md new file mode 100644 index 0000000..2a739b9 --- /dev/null +++ b/.github/copilot-instructions.md @@ -0,0 +1,41 @@ +# Copilot instructions for Scrape-UAlg-Courses + +Goal: a complete scraper + REST API for UAlg courses. Source pages are scraped into a normalized SQLite DB; the FastAPI service reads from that DB and serves a simple web UI. + +## Architecture and data flow +- Scraper (basic) in `src/scraper.py` (class `UAlgScraper`): fetches HTML and parses simple "course" blocks. Used mainly in unit tests/examples. +- Scraper (full) in `src/scrape_ualg.py` (class `UAlgCourseScraper`): + - Initializes DB from `schema.sql` via `init_db()`. + - Crawls listing pages starting at `START_URL`, extracts course links, parses each course page into fields: title, code, level, school, language, modules, areas, documents. + - Persists to SQLite (`ualg_courses.db` by default), also downloads documents to `data/docs/`. + - Respects retries (`Config.max_retries`) and rate limiting (sleep ~1.5s between courses). +- API in `src/api.py` (FastAPI): reads from the same SQLite DB (`DB_PATH = "ualg_courses.db"`) and serves endpoints (`/courses`, `/courses/{id}`, `/levels`, `/schools`, `/areas`, `/stats`, `/api`, `/`). The `/` route serves `templates/index.html`. +- Frontend in `templates/index.html`: fetches API endpoints and renders dashboards with Chart.js. + +## Dev workflows (Windows PowerShell friendly) +- Install deps: `make install`; dev deps: `make install-dev`. +- Initialize DB: `make init-db`; load demo data: `make demo` (runs `scripts/populate_demo_data.py`). +- Run scraper: `make run-scraper` (equivalent to `python -m src.scrape_ualg`). +- Run API: `make run-api` (uvicorn `src.api:app` on port 8000). +- Tests: `make test` (pytest + coverage). Lint/format: `make lint`, `make format`. + +## Conventions and patterns +- Config is centralized in `src/config.py` (`Config`): defaults to `https://www.ualg.pt`, timeout=30, max_retries=3, UA set; override with env `UALG_BASE_URL` or via constructor. +- Logging: both scrapers log progress; keep user-agent and backoff semantics; do not remove the ~1.5s delay in the full scraper. +- DB access in API: use `get_db_connection()` with `row_factory = sqlite3.Row` and helper `dict_from_row`. +- Filtering: `/courses` builds SQL dynamically with optional joins on `levels`, `schools`, and `areas`; keep `limit` (1..500) and `offset` semantics. +- Schema is authoritative: see `schema.sql`. M2M tables: `course_area`, `module_course`; uniqueness and indices are already defined. +- Paths: code locates `schema.sql` and `templates/index.html` via `..` from `src/`; keep relative paths when adding files. + +## When extending +- API: mirror existing patterns (Pydantic response models at top; small helpers; parameterized SQL; close connections). Add tags and docs to endpoints. +- Scraper: prefer adding selectors heuristically (see `parse_course_page`), and persist via existing upsert helpers; keep downloads in `data/docs/`. +- Tests: look at `tests/test_api.py`, `tests/test_scraper.py`, `tests/test_scrape_ualg.py` for expected behavior and monkeypatch patterns (e.g., override `DB_PATH` in API tests). + +## Quick examples +- Add an API filter: extend `list_courses` by appending the join + where + param; preserve `limit/offset` bounds. +- Add a new field to courses: update `schema.sql` + insert in `save_course` + select in API queries + include in models. + +Notes +- Default DB filename is `ualg_courses.db` in repo root. Populate it via scraper or `scripts/populate_demo_data.py` before using the UI. +- CI: `.github/workflows/python-app.yml` runs tests/lint on PRs; match the Makefile targets. diff --git a/.github/workflows/copilot-setup-steps.yml b/.github/workflows/copilot-setup-steps.yml new file mode 100644 index 0000000..6665e37 --- /dev/null +++ b/.github/workflows/copilot-setup-steps.yml @@ -0,0 +1,51 @@ +name: "Copilot Setup Steps" + +# Automatically run the setup steps when they are changed to allow for easy validation, +# and allow manual testing through the repository's "Actions" tab +on: + workflow_dispatch: + push: + paths: + - .github/workflows/copilot-setup-steps.yml + pull_request: + paths: + - .github/workflows/copilot-setup-steps.yml + +jobs: + # The job MUST be called copilot-setup-steps or it will not be picked up by Copilot. + copilot-setup-steps: + runs-on: ubuntu-latest + + # Set the permissions to the lowest permissions possible needed for your steps. + # Copilot will be given its own token for its operations. + permissions: + # If you want to clone the repository as part of your setup steps, for example to install dependencies, + # you'll need the `contents: read` permission. If you don't clone the repository in your setup steps, + # Copilot will do this for you automatically after the steps complete. + contents: read + + # You can define any steps you want, and they will run before the agent starts. + # If you do not check out your code, Copilot will do this for you. + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Set up Python 3.10 + uses: actions/setup-python@v5 + with: + python-version: "3.10" + cache: "pip" + + - name: Install Python dependencies + run: | + python -m pip install --upgrade pip + pip install -r requirements.txt + pip install -r requirements-dev.txt + + - name: Initialize database + run: | + python -c "from src.scrape_ualg import UAlgCourseScraper; s = UAlgCourseScraper(); s.init_db()" + + - name: Verify installation + run: | + python -c "import requests; import bs4; import fastapi; import pytest; print('All dependencies installed successfully')" diff --git a/tests/test_api.py b/tests/test_api.py index 730af93..d50db83 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -4,6 +4,7 @@ import pytest import sqlite3 +import os from fastapi.testclient import TestClient from src.api import app @@ -194,3 +195,49 @@ def test_get_stats(self, client): assert "total_modules" in data assert "total_documents" in data assert data["total_courses"] >= 2 + + def test_list_courses_with_school_filter(self, client): + """Testa listagem de cursos com filtro de escola.""" + response = client.get("/courses?school=Escola de Tecnologia") + assert response.status_code == 200 + data = response.json() + assert isinstance(data, list) + # Deve retornar cursos da escola especificada + assert len(data) >= 1 + # Verificar que todos os cursos têm uma escola associada + for course in data: + assert course.get("school_id") is not None + + def test_list_courses_with_area_filter(self, client): + """Testa listagem de cursos com filtro de área.""" + response = client.get("/courses?area=Informática") + assert response.status_code == 200 + data = response.json() + assert isinstance(data, list) + # Deve retornar cursos da área especificada + assert len(data) >= 1 + + def test_list_courses_with_offset(self, client): + """Testa listagem de cursos com paginação (offset).""" + response = client.get("/courses?offset=1&limit=1") + assert response.status_code == 200 + data = response.json() + assert isinstance(data, list) + assert len(data) <= 1 + + def test_read_root_fallback(self, client, monkeypatch): + """Testa fallback da página inicial quando index.html não existe.""" + # Mock os.path.exists para retornar False + original_exists = os.path.exists + + def mock_exists(path): + if "index.html" in path: + return False + return original_exists(path) + + monkeypatch.setattr(os.path, "exists", mock_exists) + + response = client.get("/") + assert response.status_code == 200 + assert "API de Cursos da UAlg" in response.text + assert "/docs" in response.text diff --git a/tests/test_scrape_ualg.py b/tests/test_scrape_ualg.py index abbbdce..2ccc533 100644 --- a/tests/test_scrape_ualg.py +++ b/tests/test_scrape_ualg.py @@ -209,3 +209,176 @@ def test_close(self, scraper): """Testa fechamento da sessão.""" scraper.close() # Não deve gerar exceção + + @patch("src.scrape_ualg.requests.Session.get") + def test_save_document(self, mock_get, scraper, test_db_path): + """Testa salvamento de documentos.""" + scraper.init_db() + + # Mock da resposta HTTP para download + mock_response = Mock() + mock_response.content = b"PDF content" + mock_response.raise_for_status = Mock() + mock_get.return_value = mock_response + + conn = sqlite3.connect(test_db_path) + conn.execute("PRAGMA foreign_keys = ON") + + # Inserir curso de teste + cur = conn.cursor() + cur.execute("INSERT INTO levels (name) VALUES ('Teste')") + level_id = cur.lastrowid + cur.execute("INSERT INTO courses (code, title, level_id) VALUES (?, ?, ?)", ("TEST", "Curso Teste", level_id)) + course_id = cur.lastrowid + conn.commit() + + # Salvar documento + scraper.save_document(conn, course_id, "Plano Curricular", "https://test.com/doc.pdf") + + # Verificar se foi salvo + cur.execute("SELECT * FROM course_documents WHERE course_id = ?", (course_id,)) + doc = cur.fetchone() + + assert doc is not None + conn.close() + + @patch("src.scrape_ualg.requests.Session.get") + def test_save_document_download_error(self, mock_get, scraper, test_db_path): + """Testa salvamento de documento com erro no download.""" + scraper.init_db() + + # Mock erro no download + mock_get.side_effect = requests.RequestException("Connection error") + + conn = sqlite3.connect(test_db_path) + conn.execute("PRAGMA foreign_keys = ON") + + # Inserir curso de teste + cur = conn.cursor() + cur.execute("INSERT INTO levels (name) VALUES ('Teste')") + level_id = cur.lastrowid + cur.execute("INSERT INTO courses (code, title, level_id) VALUES (?, ?, ?)", ("TEST", "Curso Teste", level_id)) + course_id = cur.lastrowid + conn.commit() + + # Tentar salvar documento (deve lidar com erro graciosamente) + scraper.save_document(conn, course_id, "Plano Curricular", "https://test.com/doc.pdf") + + # Verificar se o registro foi criado mesmo sem download + cur.execute("SELECT * FROM course_documents WHERE course_id = ?", (course_id,)) + doc = cur.fetchone() + + assert doc is not None + conn.close() + + def test_parse_course_page_with_modules(self, scraper): + """Testa parsing de página de curso com módulos.""" + html = """ + + +

Curso de Teste

+

Descrição do curso de teste

+ + + + + + + + + + + + + + + + +
CódigoDisciplinaECTS
CS101Programação I6 ECTS
CS102Algoritmos5 ECTS
+ + + """ + + soup = BeautifulSoup(html, "html.parser") + data = scraper.parse_course_page(soup, "https://test.com/curso") + + assert "modules" in data + assert len(data["modules"]) == 2 + # Verificar que os módulos específicos foram extraídos + module_titles = [m["title"] for m in data["modules"]] + assert "Programação I" in module_titles + assert "Algoritmos" in module_titles + + def test_parse_course_page_with_areas(self, scraper): + """Testa parsing de página de curso com áreas.""" + html = """ + + +

Curso de Teste

+
Informática
+
Ciências da Computação
+ + + """ + + soup = BeautifulSoup(html, "html.parser") + data = scraper.parse_course_page(soup, "https://test.com/curso") + + assert "areas" in data + assert len(data["areas"]) == 2 + # Verificar que as áreas específicas foram extraídas + assert "Informática" in data["areas"] + assert "Ciências da Computação" in data["areas"] + + @patch("src.scrape_ualg.requests.Session.get") + def test_scrape_all_courses_with_limit(self, mock_get, scraper, test_db_path): + """Testa scraping com limite de cursos.""" + scraper.init_db() + + # Mock página inicial + listing_html = """ + + + Curso 1 + Curso 2 + Curso 3 + + + """ + + course_html = """ + + +

Curso Teste

+

Descrição

+ + + """ + + mock_response = Mock() + mock_response.text = listing_html + mock_response.raise_for_status = Mock() + + mock_course_response = Mock() + mock_course_response.text = course_html + mock_course_response.raise_for_status = Mock() + + # Primeira chamada retorna listagem, demais retornam página de curso + mock_get.side_effect = [mock_response] + [mock_course_response] * 3 + + # Executar scraping com limite + scraper.scrape_all_courses(limit=2) + + # Verificar que apenas 2 cursos foram processados (além da página inicial) + assert mock_get.call_count == 3 # 1 listagem + 2 cursos + + @patch("src.scrape_ualg.requests.Session.get") + def test_scrape_all_courses_error_handling(self, mock_get, scraper, test_db_path): + """Testa tratamento de erros durante scraping.""" + scraper.init_db() + + # Mock erro na página inicial (simular exceção de rede) + mock_get.side_effect = requests.RequestException("Connection error") + + # Não deve lançar exceção + scraper.scrape_all_courses(limit=1)