diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md new file mode 100644 index 0000000..6874433 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -0,0 +1,32 @@ +--- +name: Bug report +about: Create a report to help us improve +title: '[BUG] ' +labels: bug +assignees: '' + +--- + +**Describe the bug** +A clear and concise description of what the bug is. + +**To Reproduce** +Steps to reproduce the behavior: +1. Go to '...' +2. Click on '....' +3. Scroll down to '....' +4. See error + +**Expected behavior** +A clear and concise description of what you expected to happen. + +**Screenshots** +If applicable, add screenshots to help explain your problem. + +**Environment (please complete the following information):** + - OS: [e.g. Ubuntu 20.04] + - Python Version: [e.g. 3.10] + - Project Version: [e.g. 1.0.0] + +**Additional context** +Add any other context about the problem here. diff --git a/.github/ISSUE_TEMPLATE/feature_request.md b/.github/ISSUE_TEMPLATE/feature_request.md new file mode 100644 index 0000000..0f540bf --- /dev/null +++ b/.github/ISSUE_TEMPLATE/feature_request.md @@ -0,0 +1,20 @@ +--- +name: Feature request +about: Suggest an idea for this project +title: '[FEATURE] ' +labels: enhancement +assignees: '' + +--- + +**Is your feature request related to a problem? Please describe.** +A clear and concise description of what the problem is. Ex. I'm always frustrated when [...] + +**Describe the solution you'd like** +A clear and concise description of what you want to happen. + +**Describe alternatives you've considered** +A clear and concise description of any alternative solutions or features you've considered. + +**Additional context** +Add any other context or screenshots about the feature request here. diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md new file mode 100644 index 0000000..5f19500 --- /dev/null +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -0,0 +1,36 @@ +## Description + +Please include a summary of the changes and the related issue. Please also include relevant motivation and context. + +Fixes # (issue) + +## Type of change + +Please delete options that are not relevant. + +- [ ] Bug fix (non-breaking change which fixes an issue) +- [ ] New feature (non-breaking change which adds functionality) +- [ ] Breaking change (fix or feature that would cause existing functionality to not work as expected) +- [ ] This change requires a documentation update + +## How Has This Been Tested? + +Please describe the tests that you ran to verify your changes. Provide instructions so we can reproduce. Please also list any relevant details for your test configuration. + +- [ ] Test A +- [ ] Test B + +**Test Configuration**: +* Python version: +* Operating System: + +## Checklist: + +- [ ] My code follows the style guidelines of this project +- [ ] I have performed a self-review of my own code +- [ ] I have commented my code, particularly in hard-to-understand areas +- [ ] I have made corresponding changes to the documentation +- [ ] My changes generate no new warnings +- [ ] I have added tests that prove my fix is effective or that my feature works +- [ ] New and existing unit tests pass locally with my changes +- [ ] Any dependent changes have been merged and published in downstream modules diff --git a/.github/workflows/python-app.yml b/.github/workflows/python-app.yml new file mode 100644 index 0000000..051d93a --- /dev/null +++ b/.github/workflows/python-app.yml @@ -0,0 +1,37 @@ +name: Python Application + +on: + push: + branches: [ "main", "master" ] + pull_request: + branches: [ "main", "master" ] + +permissions: + contents: read + +jobs: + build: + + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v3 + - name: Set up Python 3.10 + uses: actions/setup-python@v3 + with: + python-version: "3.10" + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install flake8 pytest + if [ -f requirements.txt ]; then pip install -r requirements.txt; fi + if [ -f requirements-dev.txt ]; then pip install -r requirements-dev.txt; fi + - name: Lint with flake8 + run: | + # stop the build if there are Python syntax errors or undefined names + flake8 . --count --select=E9,F63,F7,F82 --show-source --statistics + # exit-zero treats all errors as warnings. The GitHub editor is 127 chars wide + flake8 . --count --exit-zero --max-complexity=10 --max-line-length=127 --statistics + - name: Test with pytest + run: | + pytest diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..f4daebb --- /dev/null +++ b/.gitignore @@ -0,0 +1,134 @@ +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +pip-wheel-metadata/ +share/python-wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST + +# PyInstaller +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.nox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +*.py,cover +.hypothesis/ +.pytest_cache/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +local_settings.py +db.sqlite3 +db.sqlite3-journal + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ + +# PyBuilder +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# IPython +profile_default/ +ipython_config.py + +# pyenv +.python-version + +# pipenv +Pipfile.lock + +# PEP 582 +__pypackages__/ + +# Celery stuff +celerybeat-schedule +celerybeat.pid + +# SageMath parsed files +*.sage.py + +# Environments +.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ +.dmypy.json +dmypy.json + +# Pyre type checker +.pyre/ + +# IDE +.vscode/ +.idea/ +*.swp +*.swo +*~ + +# OS +.DS_Store +Thumbs.db diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md new file mode 100644 index 0000000..41dd9eb --- /dev/null +++ b/CODE_OF_CONDUCT.md @@ -0,0 +1,127 @@ +# Contributor Covenant Code of Conduct + +## Our Pledge + +We as members, contributors, and leaders pledge to make participation in our +community a harassment-free experience for everyone, regardless of age, body +size, visible or invisible disability, ethnicity, sex characteristics, gender +identity and expression, level of experience, education, socio-economic status, +nationality, personal appearance, race, religion, or sexual identity +and orientation. + +We pledge to act and interact in ways that contribute to an open, welcoming, +diverse, inclusive, and healthy community. + +## Our Standards + +Examples of behavior that contributes to a positive environment for our +community include: + +* Demonstrating empathy and kindness toward other people +* Being respectful of differing opinions, viewpoints, and experiences +* Giving and gracefully accepting constructive feedback +* Accepting responsibility and apologizing to those affected by our mistakes, + and learning from the experience +* Focusing on what is best not just for us as individuals, but for the + overall community + +Examples of unacceptable behavior include: + +* The use of sexualized language or imagery, and sexual attention or + advances of any kind +* Trolling, insulting or derogatory comments, and personal or political attacks +* Public or private harassment +* Publishing others' private information, such as a physical or email + address, without their explicit permission +* Other conduct which could reasonably be considered inappropriate in a + professional setting + +## Enforcement Responsibilities + +Community leaders are responsible for clarifying and enforcing our standards of +acceptable behavior and will take appropriate and fair corrective action in +response to any behavior that they deem inappropriate, threatening, offensive, +or harmful. + +Community leaders have the right and responsibility to remove, edit, or reject +comments, commits, code, wiki edits, issues, and other contributions that are +not aligned to this Code of Conduct, and will communicate reasons for moderation +decisions when appropriate. + +## Scope + +This Code of Conduct applies within all community spaces, and also applies when +an individual is officially representing the community in public spaces. +Examples of representing our community include using an official e-mail address, +posting via an official social media account, or acting as an appointed +representative at an online or offline event. + +## Enforcement + +Instances of abusive, harassing, or otherwise unacceptable behavior may be +reported to the community leaders responsible for enforcement. +All complaints will be reviewed and investigated promptly and fairly. + +All community leaders are obligated to respect the privacy and security of the +reporter of any incident. + +## Enforcement Guidelines + +Community leaders will follow these Community Impact Guidelines in determining +the consequences for any action they deem in violation of this Code of Conduct: + +### 1. Correction + +**Community Impact**: Use of inappropriate language or other behavior deemed +unprofessional or unwelcome in the community. + +**Consequence**: A private, written warning from community leaders, providing +clarity around the nature of the violation and an explanation of why the +behavior was inappropriate. A public apology may be requested. + +### 2. Warning + +**Community Impact**: A violation through a single incident or series +of actions. + +**Consequence**: A warning with consequences for continued behavior. No +interaction with the people involved, including unsolicited interaction with +those enforcing the Code of Conduct, for a specified period of time. This +includes avoiding interactions in community spaces as well as external channels +like social media. Violating these terms may lead to a temporary or +permanent ban. + +### 3. Temporary Ban + +**Community Impact**: A serious violation of community standards, including +sustained inappropriate behavior. + +**Consequence**: A temporary ban from any sort of interaction or public +communication with the community for a specified period of time. No public or +private interaction with the people involved, including unsolicited interaction +with those enforcing the Code of Conduct, is allowed during this period. +Violating these terms may lead to a permanent ban. + +### 4. Permanent Ban + +**Community Impact**: Demonstrating a pattern of violation of community +standards, including sustained inappropriate behavior, harassment of an +individual, or aggression toward or disparagement of classes of individuals. + +**Consequence**: A permanent ban from any sort of public interaction within +the community. + +## Attribution + +This Code of Conduct is adapted from the [Contributor Covenant][homepage], +version 2.0, available at +https://www.contributor-covenant.org/version/2/0/code_of_conduct.html. + +Community Impact Guidelines were inspired by [Mozilla's code of conduct +enforcement ladder](https://github.com/mozilla/diversity). + +[homepage]: https://www.contributor-covenant.org + +For answers to common questions about this code of conduct, see the FAQ at +https://www.contributor-covenant.org/faq. Translations are available at +https://www.contributor-covenant.org/translations. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..23a9558 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,79 @@ +# Contributing to Scrape-UAlg-Courses + +First off, thank you for considering contributing to Scrape-UAlg-Courses! It's people like you that make this project better. + +## Code of Conduct + +This project and everyone participating in it is governed by our Code of Conduct. By participating, you are expected to uphold this code. + +## How Can I Contribute? + +### Reporting Bugs + +Before creating bug reports, please check the issue list as you might find out that you don't need to create one. When you are creating a bug report, please include as many details as possible using the bug report template. + +### Suggesting Enhancements + +Enhancement suggestions are tracked as GitHub issues. When you are creating an enhancement suggestion, please include as many details as possible using the feature request template. + +### Pull Requests + +1. Fork the repository +2. Create a new branch from `main` for your feature or bugfix +3. Make your changes following our coding standards +4. Add or update tests as needed +5. Ensure all tests pass +6. Update documentation if needed +7. Submit a pull request + +## Development Setup + +1. Clone your fork: +```bash +git clone https://github.com/your-username/Scrape-UAlg-Courses.git +cd Scrape-UAlg-Courses +``` + +2. Install dependencies: +```bash +make install-dev +``` + +3. Run tests: +```bash +make test +``` + +4. Run linter: +```bash +make lint +``` + +## Coding Standards + +- Follow PEP 8 style guide +- Write meaningful commit messages +- Add docstrings to all functions and classes +- Write tests for new functionality +- Keep functions small and focused +- Use type hints where appropriate + +## Testing + +- Write unit tests for all new functionality +- Ensure all tests pass before submitting PR +- Aim for high test coverage +- Use meaningful test names that describe what is being tested + +## Documentation + +- Update README.md if you change functionality +- Add docstrings to all public functions and classes +- Comment complex logic +- Update CHANGELOG.md with notable changes + +## Questions? + +Feel free to open an issue with your question or reach out to the maintainers. + +Thank you for your contribution! 🎉 diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..ee34453 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2024 Monynha Softwares + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..11ddd7f --- /dev/null +++ b/Makefile @@ -0,0 +1,39 @@ +.PHONY: help install install-dev test lint format clean run + +help: + @echo "Available targets:" + @echo " install - Install production dependencies" + @echo " install-dev - Install development dependencies" + @echo " test - Run tests with pytest" + @echo " lint - Run linting with flake8" + @echo " format - Format code with black" + @echo " clean - Remove build artifacts and cache files" + @echo " run - Run the scraper" + +install: + pip install -r requirements.txt + +install-dev: + pip install -r requirements.txt + pip install -r requirements-dev.txt + +test: + pytest tests/ -v --cov=src --cov-report=html --cov-report=term + +lint: + flake8 src/ tests/ --max-line-length=127 --exclude=__pycache__ + +format: + black src/ tests/ --line-length=127 + +clean: + find . -type d -name "__pycache__" -exec rm -rf {} + + find . -type f -name "*.pyc" -delete + find . -type f -name "*.pyo" -delete + find . -type d -name "*.egg-info" -exec rm -rf {} + + find . -type d -name ".pytest_cache" -exec rm -rf {} + + find . -type f -name ".coverage" -delete + find . -type d -name "htmlcov" -exec rm -rf {} + + +run: + python -m src.scraper diff --git a/README.md b/README.md index 3be4402..3679c6c 100644 --- a/README.md +++ b/README.md @@ -1 +1,206 @@ -# Scrape-UAlg-Courses \ No newline at end of file +# Scrape-UAlg-Courses + +A Python-based web scraper for collecting course information from the University of Algarve (UAlg) website. + +## Features + +- 🚀 Easy-to-use API for scraping UAlg course data +- 🔧 Configurable scraper settings (timeouts, retries, user agents) +- 📦 Modular design with separate configuration and scraping logic +- ✅ Comprehensive test coverage +- 🔄 Automatic retry mechanism for failed requests +- 📝 Detailed logging for debugging + +## Installation + +### Prerequisites + +- Python 3.10 or higher +- pip (Python package manager) + +### Setup + +1. Clone the repository: +```bash +git clone https://github.com/Monynha-Softwares/Scrape-UAlg-Courses.git +cd Scrape-UAlg-Courses +``` + +2. Install dependencies: +```bash +make install +``` + +Or manually: +```bash +pip install -r requirements.txt +``` + +### Development Setup + +For development, install additional dependencies: +```bash +make install-dev +``` + +Or manually: +```bash +pip install -r requirements-dev.txt +``` + +## Usage + +### Basic Usage + +```python +from src.scraper import UAlgScraper +from src.config import Config + +# Using default configuration +scraper = UAlgScraper() +courses = scraper.scrape_courses() + +for course in courses: + print(f"{course['code']}: {course['name']}") + +scraper.close() +``` + +### Using Custom Configuration + +```python +from src.scraper import UAlgScraper +from src.config import Config + +# Custom configuration +config = Config( + base_url="https://www.ualg.pt", + timeout=60, + max_retries=5 +) + +scraper = UAlgScraper(config) +courses = scraper.scrape_courses() +scraper.close() +``` + +### Using Context Manager + +```python +from src.scraper import UAlgScraper + +with UAlgScraper() as scraper: + courses = scraper.scrape_courses() + # Process courses... +``` + +## Project Structure + +``` +monynha-backend-scraper/ +├── .github/ +│ ├── workflows/ +│ │ └── python-app.yml # GitHub Actions CI/CD +│ ├── ISSUE_TEMPLATE/ +│ │ ├── bug_report.md # Bug report template +│ │ └── feature_request.md # Feature request template +│ └── PULL_REQUEST_TEMPLATE.md # PR template +├── src/ +│ ├── __init__.py # Package initialization +│ ├── scraper.py # Main scraper logic +│ └── config.py # Configuration management +├── tests/ +│ ├── __init__.py +│ └── test_scraper.py # Unit tests +├── .gitignore # Git ignore rules +├── README.md # This file +├── LICENSE # MIT License +├── CONTRIBUTING.md # Contribution guidelines +├── CODE_OF_CONDUCT.md # Code of conduct +├── requirements.txt # Production dependencies +├── requirements-dev.txt # Development dependencies +└── Makefile # Common tasks automation +``` + +## Development + +### Running Tests + +Run all tests: +```bash +make test +``` + +Or using pytest directly: +```bash +pytest tests/ -v +``` + +### Linting + +Check code style: +```bash +make lint +``` + +### Formatting + +Format code with black: +```bash +make format +``` + +### Cleaning + +Remove build artifacts and cache files: +```bash +make clean +``` + +## Configuration + +The scraper can be configured using the `Config` class or environment variables: + +### Configuration Options + +- `base_url`: Base URL for UAlg website (default: `https://www.ualg.pt`) +- `timeout`: Request timeout in seconds (default: 30) +- `user_agent`: User agent string for requests +- `max_retries`: Maximum number of retry attempts (default: 3) + +### Environment Variables + +- `UALG_BASE_URL`: Override the default base URL + +## Contributing + +Contributions are welcome! Please read [CONTRIBUTING.md](CONTRIBUTING.md) for details on our code of conduct and the process for submitting pull requests. + +## License + +This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details. + +## Code of Conduct + +Please read [CODE_OF_CONDUCT.md](CODE_OF_CONDUCT.md) for details on our code of conduct. + +## Support + +If you encounter any issues or have questions: + +1. Check the [Issues](https://github.com/Monynha-Softwares/Scrape-UAlg-Courses/issues) page +2. Create a new issue using the appropriate template +3. Provide as much detail as possible + +## Acknowledgments + +- Built with [Requests](https://requests.readthedocs.io/) and [Beautiful Soup](https://www.crummy.com/software/BeautifulSoup/) +- Inspired by the need for easy access to UAlg course information + +## Roadmap + +- [ ] Add support for more data fields +- [ ] Implement caching mechanism +- [ ] Add CLI interface +- [ ] Create data export functionality (JSON, CSV) +- [ ] Add scheduling support for periodic scraping \ No newline at end of file diff --git a/requirements-dev.txt b/requirements-dev.txt new file mode 100644 index 0000000..5596cd4 --- /dev/null +++ b/requirements-dev.txt @@ -0,0 +1,6 @@ +pytest>=7.4.0 +pytest-cov>=4.1.0 +pytest-mock>=3.11.0 +flake8>=6.0.0 +black>=23.7.0 +mypy>=1.4.0 diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..4cca127 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,3 @@ +requests>=2.31.0 +beautifulsoup4>=4.12.0 +lxml>=4.9.0 diff --git a/src/__init__.py b/src/__init__.py new file mode 100644 index 0000000..901303f --- /dev/null +++ b/src/__init__.py @@ -0,0 +1,14 @@ +""" +UAlg Courses Scraper Package + +This package provides functionality for scraping course information +from the University of Algarve (UAlg) website. +""" + +__version__ = "0.1.0" +__author__ = "Monynha Softwares" + +from .scraper import UAlgScraper +from .config import Config + +__all__ = ["UAlgScraper", "Config"] diff --git a/src/config.py b/src/config.py new file mode 100644 index 0000000..5dc6c90 --- /dev/null +++ b/src/config.py @@ -0,0 +1,49 @@ +""" +Configuration module for UAlg scraper. + +This module handles all configuration settings for the scraper, +including URLs, timeouts, and other parameters. +""" + +import os +from typing import Optional + + +class Config: + """Configuration class for UAlg scraper.""" + + def __init__( + self, + base_url: Optional[str] = None, + timeout: int = 30, + user_agent: Optional[str] = None, + max_retries: int = 3 + ): + """ + Initialize configuration. + + Args: + base_url: Base URL for UAlg website. Defaults to environment variable or default URL. + timeout: Request timeout in seconds. Default is 30. + user_agent: User agent string. Defaults to a standard browser UA. + max_retries: Maximum number of retry attempts. Default is 3. + """ + self.base_url = base_url or os.getenv( + "UALG_BASE_URL", + "https://www.ualg.pt" + ) + self.timeout = timeout + self.user_agent = user_agent or ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/120.0.0.0 Safari/537.36" + ) + self.max_retries = max_retries + + def __repr__(self) -> str: + """String representation of Config.""" + return ( + f"Config(base_url='{self.base_url}', " + f"timeout={self.timeout}, " + f"max_retries={self.max_retries})" + ) diff --git a/src/scraper.py b/src/scraper.py new file mode 100644 index 0000000..0dedac7 --- /dev/null +++ b/src/scraper.py @@ -0,0 +1,130 @@ +""" +Main scraper module for UAlg courses. + +This module provides the UAlgScraper class for scraping course information +from the University of Algarve website. +""" + +import requests +from typing import List, Dict, Optional +from bs4 import BeautifulSoup +import logging + +from .config import Config + +# Set up logging - only configure if not already configured +if not logging.getLogger().handlers: + logging.basicConfig( + level=logging.INFO, + format='%(asctime)s - %(name)s - %(levelname)s - %(message)s' + ) +logger = logging.getLogger(__name__) + + +class UAlgScraper: + """Scraper for UAlg course information.""" + + def __init__(self, config: Optional[Config] = None): + """ + Initialize the scraper. + + Args: + config: Configuration object. If None, uses default configuration. + """ + self.config = config or Config() + self.session = requests.Session() + self.session.headers.update({ + 'User-Agent': self.config.user_agent + }) + logger.info(f"Initialized UAlgScraper with base URL: {self.config.base_url}") + + def fetch_page(self, url: str) -> Optional[str]: + """ + Fetch a web page. + + Args: + url: URL to fetch. + + Returns: + Page content as string, or None if fetch fails. + """ + for attempt in range(self.config.max_retries): + try: + logger.info(f"Fetching {url} (attempt {attempt + 1}/{self.config.max_retries})") + response = self.session.get( + url, + timeout=self.config.timeout + ) + response.raise_for_status() + logger.info(f"Successfully fetched {url}") + return response.text + except requests.RequestException as e: + logger.warning(f"Attempt {attempt + 1} failed: {e}") + if attempt == self.config.max_retries - 1: + logger.error(f"Failed to fetch {url} after {self.config.max_retries} attempts") + return None + return None + + def parse_courses(self, html: str) -> List[Dict[str, str]]: + """ + Parse course information from HTML. + + Args: + html: HTML content to parse. + + Returns: + List of dictionaries containing course information. + """ + soup = BeautifulSoup(html, 'html.parser') + courses = [] + + # This is a placeholder implementation + # Actual parsing logic would depend on the website structure + course_elements = soup.find_all('div', class_='course') + + for element in course_elements: + course = { + 'name': element.find('h2').text.strip() if element.find('h2') else '', + 'code': element.find('span', class_='code').text.strip() if element.find('span', class_='code') else '', + 'description': (element.find('p', class_='description').text.strip() + if element.find('p', class_='description') else '') + } + courses.append(course) + + logger.info(f"Parsed {len(courses)} courses") + return courses + + def scrape_courses(self, courses_url: Optional[str] = None) -> List[Dict[str, str]]: + """ + Scrape courses from UAlg website. + + Args: + courses_url: URL to scrape courses from. If None, uses base URL. + + Returns: + List of dictionaries containing course information. + """ + url = courses_url or f"{self.config.base_url}/courses" + logger.info(f"Starting course scraping from {url}") + + html = self.fetch_page(url) + if html is None: + logger.error("Failed to fetch courses page") + return [] + + courses = self.parse_courses(html) + logger.info(f"Scraping completed. Found {len(courses)} courses") + return courses + + def close(self): + """Close the session.""" + self.session.close() + logger.info("Session closed") + + def __enter__(self): + """Context manager entry.""" + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + """Context manager exit.""" + self.close() diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..46816dd --- /dev/null +++ b/tests/__init__.py @@ -0,0 +1 @@ +"""Tests package.""" diff --git a/tests/test_scraper.py b/tests/test_scraper.py new file mode 100644 index 0000000..5a766c5 --- /dev/null +++ b/tests/test_scraper.py @@ -0,0 +1,159 @@ +""" +Unit tests for the UAlg scraper. + +This module contains tests for the scraper functionality. +""" + +from unittest.mock import Mock, patch +import requests + +from src.scraper import UAlgScraper +from src.config import Config + + +class TestConfig: + """Test cases for Config class.""" + + def test_config_default_values(self): + """Test that Config initializes with default values.""" + config = Config() + assert config.base_url == "https://www.ualg.pt" + assert config.timeout == 30 + assert config.max_retries == 3 + assert config.user_agent is not None + + def test_config_custom_values(self): + """Test that Config accepts custom values.""" + config = Config( + base_url="https://example.com", + timeout=60, + max_retries=5 + ) + assert config.base_url == "https://example.com" + assert config.timeout == 60 + assert config.max_retries == 5 + + def test_config_repr(self): + """Test Config string representation.""" + config = Config() + repr_str = repr(config) + assert "Config" in repr_str + assert "https://www.ualg.pt" in repr_str + + +class TestUAlgScraper: + """Test cases for UAlgScraper class.""" + + def test_scraper_initialization(self): + """Test scraper initialization.""" + scraper = UAlgScraper() + assert scraper.config is not None + assert scraper.session is not None + + def test_scraper_with_custom_config(self): + """Test scraper with custom configuration.""" + config = Config(base_url="https://test.com") + scraper = UAlgScraper(config) + assert scraper.config.base_url == "https://test.com" + + @patch('src.scraper.requests.Session.get') + def test_fetch_page_success(self, mock_get): + """Test successful page fetch.""" + mock_response = Mock() + mock_response.text = "
Test" + mock_response.raise_for_status = Mock() + mock_get.return_value = mock_response + + scraper = UAlgScraper() + result = scraper.fetch_page("https://example.com") + + assert result == "Test" + mock_get.assert_called_once() + + @patch('src.scraper.requests.Session.get') + def test_fetch_page_failure(self, mock_get): + """Test page fetch failure with retries.""" + mock_get.side_effect = requests.RequestException("Connection error") + + config = Config(max_retries=2) + scraper = UAlgScraper(config) + result = scraper.fetch_page("https://example.com") + + assert result is None + assert mock_get.call_count == 2 + + def test_parse_courses_empty(self): + """Test parsing empty HTML.""" + scraper = UAlgScraper() + html = "" + courses = scraper.parse_courses(html) + assert courses == [] + + def test_parse_courses_with_data(self): + """Test parsing HTML with course data.""" + scraper = UAlgScraper() + html = """ + + +Introduction to CS
+Calculus I
+Introduction to Physics
+