From 5c2f934b5655d87a9bbe83f39d3c06c993ba75d9 Mon Sep 17 00:00:00 2001 From: Lichenchao <517024110@qq.com> Date: Fri, 10 Oct 2025 00:20:56 +0800 Subject: [PATCH 01/13] =?UTF-8?q?feat(fancontroller):=20=E5=AE=9E=E7=8E=B0?= =?UTF-8?q?=E9=82=AE=E4=BB=B6=E5=91=8A=E8=AD=A6=E5=92=8C=E5=8D=95=E6=AC=A1?= =?UTF-8?q?=E6=89=A7=E8=A1=8C=E6=A8=A1=E5=BC=8F-=20=E6=B7=BB=E5=8A=A0?= =?UTF-8?q?=E9=82=AE=E4=BB=B6=E5=91=8A=E8=AD=A6=E5=8A=9F=E8=83=BD=EF=BC=8C?= =?UTF-8?q?=E6=94=AF=E6=8C=81=E9=A3=8E=E6=89=87=E5=BC=82=E5=B8=B8=E6=97=B6?= =?UTF-8?q?=E8=87=AA=E5=8A=A8=E5=8F=91=E9=80=81=E5=91=8A=E8=AD=A6=E9=82=AE?= =?UTF-8?q?=E4=BB=B6-=20=E5=AE=9E=E7=8E=B0=E5=8D=95=E6=AC=A1=E6=89=A7?= =?UTF-8?q?=E8=A1=8C=E6=A8=A1=E5=BC=8F=EF=BC=8C=E6=94=AF=E6=8C=81=E8=A2=AB?= =?UTF-8?q?=E5=A4=96=E9=83=A8=E8=B0=83=E5=BA=A6=E5=B7=A5=E5=85=B7=E8=B0=83?= =?UTF-8?q?=E7=94=A8=20-=E9=87=8D=E6=9E=84=E9=A3=8E=E6=89=87=E6=8E=A7?= =?UTF-8?q?=E5=88=B6=E9=80=BB=E8=BE=91=EF=BC=8C=E5=88=86=E7=A6=BB=E5=BE=AA?= =?UTF-8?q?=E7=8E=AF=E5=92=8C=E5=8D=95=E6=AC=A1=E6=89=A7=E8=A1=8C=E6=A8=A1?= =?UTF-8?q?=E5=BC=8F=20-=20=E6=B7=BB=E5=8A=A0=E8=AF=A6=E7=BB=86=E7=9A=84?= =?UTF-8?q?=E8=BF=90=E8=A1=8C=E6=97=B6=E7=BB=9F=E8=AE=A1=E5=92=8C=E6=80=A7?= =?UTF-8?q?=E8=83=BD=E7=9B=91=E6=8E=A7=20-=20=E6=94=AF=E6=8C=81=20YAML=20?= =?UTF-8?q?=E9=85=8D=E7=BD=AE=E6=96=87=E4=BB=B6=E6=A0=BC=E5=BC=8F=20-=20?= =?UTF-8?q?=E4=BC=98=E5=8C=96=20Dell=20730=20=E6=9C=8D=E5=8A=A1=E5=99=A8?= =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E6=B5=81=E7=A8=8B-=20=E6=B7=BB?= =?UTF-8?q?=E5=8A=A0=E6=B8=A9=E5=BA=A6=E5=8F=98=E5=8C=96=E8=B6=8B=E5=8A=BF?= =?UTF-8?q?=E5=88=86=E6=9E=90=E5=92=8C=E9=97=B4=E9=9A=94=E7=BB=9F=E8=AE=A1?= =?UTF-8?q?=20-=20=E5=AE=9E=E7=8E=B0=E5=A4=B1=E8=B4=A5=E6=AC=A1=E6=95=B0?= =?UTF-8?q?=E7=BB=9F=E8=AE=A1=E5=92=8C=E5=91=8A=E8=AD=A6=E9=98=B2=E6=8A=96?= =?UTF-8?q?=E6=9C=BA=E5=88=B6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .claude/settings.local.json | 10 + .gitignore | 1 + CLAUDE.md | 164 ++++++++++++++++ README.md | 279 +++++++++++++++++++--------- README_EN.md | 152 ++++++++++----- fanController/base_controller.py | 252 ++++++++++++++++++++----- fanController/dell730_controller.py | 70 ++++++- fan_settings.json.template | 62 ------- fan_settings.yaml.template | 84 +++++++++ fancontroller.py | 46 +++-- fancontroller_once.py | 97 ++++++++++ requirements.txt | 1 + utils/__init__.py | 7 + utils/email_notifier.py | 156 ++++++++++++++++ 14 files changed, 1120 insertions(+), 261 deletions(-) create mode 100644 .claude/settings.local.json create mode 100644 CLAUDE.md delete mode 100644 fan_settings.json.template create mode 100644 fan_settings.yaml.template create mode 100644 fancontroller_once.py create mode 100644 utils/__init__.py create mode 100644 utils/email_notifier.py diff --git a/.claude/settings.local.json b/.claude/settings.local.json new file mode 100644 index 0000000..33d9fc9 --- /dev/null +++ b/.claude/settings.local.json @@ -0,0 +1,10 @@ +{ + "permissions": { + "allow": [ + "Bash(python:*)", + "Bash(pip install:*)" + ], + "deny": [], + "ask": [] + } +} diff --git a/.gitignore b/.gitignore index 61faa3e..c86d4cb 100644 --- a/.gitignore +++ b/.gitignore @@ -164,4 +164,5 @@ cython_debug/ # Custom ignores fan_settings.json fancontroller.log +fan_settings.yaml logs/ \ No newline at end of file diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..00a9dbc --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,164 @@ +# CLAUDE.md + +This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. + +## 项目概述 + +这是一个跨平台的 IPMI 风扇控制器,用于监控服务器 CPU 温度并根据预定义的温度区间自动调整风扇转速。支持 Windows 和 Linux 平台。 + +### 两种运行模式 + +1. **循环控制模式** (`fancontroller.py`): 程序内部循环监控,适合长期后台运行 +2. **单次执行模式** (`fancontroller_once.py`): 执行一次后退出,适合外部调度工具(cron、systemd timer)定时调用 + +## 核心架构 + +### 三层控制器架构 +- **fancontroller.py**: 循环模式入口,负责配置加载、日志初始化和多线程管理 +- **fancontroller_once.py**: 单次执行模式入口,执行一次后退出 +- **base_controller.py**: `IPMIFanController` 基类,定义通用的 IPMI 命令执行逻辑 + - `adjust_fans_once()`: 核心方法,执行一次温度检测和风扇调整 + - `process_server_loop()`: 循环模式,持续调用 `adjust_fans_once()` + - `process_server_once()`: 单次模式,调用一次 `adjust_fans_once()` 后退出 +- **dell730_controller.py**: `Dell730FanController` 子类,实现 Dell 730 系列服务器的具体控制逻辑 + - `_initialize_dell730()`: Dell 730 特定初始化(禁用 PCIe 散热响应、设置手动模式) + - `start_fan_control()`: 启动循环模式 + - `run_once()`: 执行单次模式 + +### 多线程模型 +- 每个服务器实例在独立线程中运行 +- 线程命名格式: `Thread-{server_ip}` +- 主线程等待所有子线程完成 (`thread.join()`) + +### 配置系统 +- 使用 YAML 格式配置: `fan_settings.yaml` (从 `fan_settings.yaml.template` 复制) +- 支持多服务器配置,每个服务器可定义多个温度区间和对应风扇转速 +- IP 地址可设置为 `"local"` 以在本地直接执行命令(无需远程 IPMI) +- **邮件告警** (可选,默认关闭): 风扇调节连续失败时自动发送告警邮件 + - 可配置转速阈值和失败次数 + - 支持多个收件人 + - 防止邮件轰炸(最小间隔 1 小时) + +### 日志系统 +- 使用 `TimedRotatingFileHandler` 按天轮转日志 +- 日志保留天数由配置文件中的 `log_backup_count` 控制 +- 同时输出到文件 (`logs/fancontroller.log`) 和控制台 +- 日志格式: `时间戳 - 线程名 - 消息内容` + +### 温度监控逻辑 +- 使用 `max(cpu_temps)` 作为判断依据(最热 CPU 温度) +- 温度范围匹配: `min_temp <= avg_temp < max_temp` +- 优化策略: 温度未跨越区间且风扇未异常飙升时跳过设置 +- 风扇异常检测: `max_speed >= 15000` (接近 `max_fan_rotational_speed = 16000`) + +## 常用开发命令 + +### 安装依赖 +```bash +pip install -r requirements.txt +``` + +### 配置文件准备 +```bash +# 复制模板文件 +cp fan_settings.yaml.template fan_settings.yaml + +# 编辑配置文件,设置服务器 IP、用户名、密码和温度区间 +# 注意: 只能使用 IP 地址,不能使用域名 +``` + +### 运行程序 + +**循环控制模式(程序内部循环)** +```bash +# 前台运行 +python fancontroller.py + +# 后台运行(Windows) +start /b python fancontroller.py + +# 后台运行(Linux) +nohup python3 fancontroller.py & +``` + +**单次执行模式(外部调度)** +```bash +# 直接执行一次 +python fancontroller_once.py + +# cron 定时执行(每 10 分钟) +*/10 * * * * /usr/bin/python3 /path/to/fancontroller_once.py + +# Windows 任务计划程序 +schtasks /create /tn "IPMI Fan Controller" /tr "python C:\path\to\fancontroller_once.py" /sc minute /mo 10 +``` + +**作为 systemd 服务运行 (Linux 推荐)** +```bash +# 创建服务文件 +sudo nano /etc/systemd/system/fancontroller.service + +# 重载配置并启动服务 +sudo systemctl daemon-reload +sudo systemctl start fancontroller.service +sudo systemctl status fancontroller.service +sudo systemctl enable fancontroller.service + +# 查看服务日志 +journalctl -u fancontroller.service -f +``` + +### 查看日志 +```bash +# 实时查看日志 +tail -f logs/fancontroller.log + +# 查看历史日志 +ls logs/ +cat logs/fancontroller.log.YYYY-MM-DD +``` + +## 添加新服务器型号支持 + +1. 在 `fanController/` 目录下创建新的控制器类文件 (如 `dellr410_controller.py`) +2. 继承 `IPMIFanController` 基类并实现以下方法: + - `get_cpu_temperature()`: 解析 IPMI 温度传感器输出 + - `set_fan_speed(fan_index, percentage)`: 设置风扇转速的 IPMI raw 命令 + - `get_fan_rotational_speed()`: 解析 IPMI 风扇转速输出 + - `process_server_loop()`: 重写循环模式(如需特定初始化) + - `process_server_once()`: 重写单次模式(如需特定初始化) + - `start_fan_control()`: 启动循环控制 + - `run_once()`: 执行单次控制 +3. (可选) 重写 `set_ipmi_manual_mode()` 和 `set_ipmi_auto_mode()` 如果命令不同 +4. 在 `fancontroller.py` 和 `fancontroller_once.py` 中添加新类型的判断逻辑 + +## 重要注意事项 + +### IPMI 命令执行 +- Windows 平台使用配置文件中的 `windows_ipmi_tool_path` 指定的 ipmitool.exe +- Linux 平台直接使用系统的 `ipmitool` 命令(需预先安装) +- 本地模式 (`ip: "local"`): 命令格式无需 `-I lanplus -H ...` 前缀 + +### Dell 730 特定细节 + +**初始化命令 (必须)** +- 禁用"第三方 PCIe 卡的散热响应策略": `raw 0x30 0xce 0x00 0x16 0x05 0x00 0x00 0x00 0x05 0x00 0x01 0x00 0x00` +- 成功返回: `16 05 00 00 00` +- 此命令需要在设置手动模式前执行,否则风扇控制可能不生效 + +**IPMI 命令** +- CPU 温度传感器识别: `"0Eh"` 或 `"0Fh"` 标识符 +- 手动模式 IPMI 命令: `raw 0x30 0x30 0x01 0x00` +- 自动模式 IPMI 命令: `raw 0x30 0x30 0x01 0x01` +- 设置风扇转速: `raw 0x30 0x30 0x02 0x{fan_index:02x} 0x{percentage:02x}` + +### 依赖项 +- **PyYAML**: 用于解析 YAML 配置文件 +- 其他功能仅依赖 Python 标准库 + +## 兼容的服务器型号 + +| 品牌 | 型号 | Type 配置值 | +|------|------|------------| +| Dell | 730XD | `dell730` | +| Dell | 730 | `dell730` | diff --git a/README.md b/README.md index c43e651..d08eb71 100644 --- a/README.md +++ b/README.md @@ -32,97 +32,208 @@ ``` cd python-ipmitool ``` -3. 复制 `fan_settings.json.template` 文件为 `fan_settings.json`。 -4. 编辑新创建的 `fan_settings.json` 配置文件,其含义如下,需要自己配置ip地址和风扇转速 +3. 安装依赖 + + ``` + pip install -r requirements.txt + ``` + +4. 复制 `fan_settings.yaml.template` 文件为 `fan_settings.yaml`。 + + ```bash + # Linux/Mac + cp fan_settings.yaml.template fan_settings.yaml + + # Windows + copy fan_settings.yaml.template fan_settings.yaml + ``` + +5. 编辑新创建的 `fan_settings.yaml` 配置文件,其含义如下,需要自己配置ip地址和风扇转速 > 注意只能用ip地址,不能用域名 > - > 注意,添加或删除服务配置的时候,要注意{}不要带多余的逗号,否则会报错json.decoder.JSONDecodeError + > **新功能**: 支持邮件告警,当风扇调节连续失败时自动发送告警邮件(默认关闭) > + ```yaml + # IPMI 风扇控制器配置文件 + + # 是否自动控制风扇转速,true 为自动控制,false 为手动控制 + auto: true + + # 控制风扇转速的时间间隔,单位为秒 + interval: 60 + + # 日志文件保留天数 + log_backup_count: 30 + + # Windows 系统下 ipmitool 工具的路径 + windows_ipmi_tool_path: ".\\ipmitool\\ipmitool.exe" + + # 告警配置(可选,默认关闭) + alert: + enabled: false # 是否启用邮件告警,默认 false + fan_speed_threshold: 10000 # 风扇转速异常阈值(RPM) + max_failed_attempts: 3 # 连续失败次数阈值 + email: + smtp_server: "smtp.gmail.com" # SMTP 服务器 + smtp_port: 587 # SMTP 端口 + use_tls: true # 是否使用 TLS + sender_email: "your_email@gmail.com" + sender_password: "your_app_password" + recipient_emails: + - "admin@example.com" + + # 服务器列表 + servers: + - type: dell730 # 服务器类型 + ip: "192.168.71.90" # 服务器 IP 地址。如果脚本与服务器在同一台机器运行,可设置为 "local" 以直接执行本地命令 + user: root # IPMI 用户名 + password: "123123" # IPMI 密码 + temperature_ranges: # 温度范围与对应的风扇转速 + - min_temp: 0 # 区间最低温度(包括) + max_temp: 60 # 区间最高温度(包括) + fan_speeds: [20, 20, 20, 20, 20, 20] # 对应风扇转速的列表,单位为百分比 + - min_temp: 61 + max_temp: 80 + fan_speeds: [25, 25, 25, 25, 25, 25] + + - type: dell730 + ip: "192.168.71.91" + user: root + password: "123123" + temperature_ranges: + - min_temp: 0 + max_temp: 60 + fan_speeds: [20, 20, 20, 20, 20, 20] + - min_temp: 61 + max_temp: 80 + fan_speeds: [25, 25, 25, 25, 25, 25] + + - type: dell730 + ip: "192.168.71.92" + user: root + password: "123123" + temperature_ranges: + - min_temp: 0 + max_temp: 60 + fan_speeds: [20, 20, 20, 20, 20, 20] + - min_temp: 61 + max_temp: 80 + fan_speeds: [25, 25, 25, 25, 25, 25] ``` - { - "auto": true, // 是否自动控制风扇转速,true为自动控制,false为手动控制 - "interval": 60, // 控制风扇转速的时间间隔,单位为秒 - "log_backup_count": 30, // 日志文件保留天数 - "windows_ipmi_tool_path": ".\\ipmitool\\ipmitool.exe", // Windows 系统下 ipmitool 工具的路径 - "servers": [ // 服务器列表 - { - "type": "dell730", // 服务器类型 - "ip": "192.168.71.90", // 服务器 IP 地址。如果脚本与服务器在同一台机器运行,可设置为 "local" 以直接执行本地命令。 - "user": "root", // IPMI 用户名 - "password": "123123", // IPMI 密码 - "temperature_ranges": [ // 温度范围与对应的风扇转速 - { - "min_temp": 0, // 区间最低温度(包括) - "max_temp": 60, // 区间最高温度(包括) - "fan_speeds": [20,20,20,20,20,20] // 对应风扇转速的列表,每个元素表示一个风扇的转速,单位为百分比 - }, - { - "min_temp": 61, // 区间最低温度(包括) - "max_temp": 80, // 区间最高温度(包括) - "fan_speeds": [25,25,25,25,25,25] // 对应风扇转速的列表,每个元素表示一个风扇的转速,单位为百分比 - } - ] - }, - { - "type": "dell730", - "ip": "192.168.71.91", - "user": "root", - "password": "123123", - "temperature_ranges": [ - { - "min_temp": 0, - "max_temp": 60, - "fan_speeds": [20,20,20,20,20,20] - }, - { - "min_temp": 61, - "max_temp": 80, - "fan_speeds": [25,25,25,25,25,25] - } - ] - }, - { - "type": "dell730", - "ip": "192.168.71.92", - "user": "root", - "password": "123123", - "temperature_ranges": [ - { - "min_temp": 0, - "max_temp": 60, - "fan_speeds": [20,20,20,20,20,20] - }, - { - "min_temp": 61, - "max_temp": 80, - "fan_speeds": [25,25,25,25,25,25] - } - ] - } - ] - } - - ``` -5. 启动项目 - - 为了方便长期运行,推荐采用后台运行的方式。 - - 1. **Windows 环境** - - 使用 `start /b` 命令让脚本在后台运行: - ``` - start /b python fancontroller.py - ``` - - 2. **Linux 环境** - - 使用 `nohup` 和 `&` 让脚本在后台运行,并确保退出终端后进程不被终止: - ``` - nohup python3 fancontroller.py & - ``` + + +6. 启动项目 + + 项目提供两种运行模式: + + ## 模式一:循环控制模式(推荐用于长期运行) + + 程序内部循环监控温度并调整风扇,适合作为后台服务运行。 + + **前台运行(调试用)** + ```bash + # Windows + python fancontroller.py + + # Linux + python3 fancontroller.py + ``` + + **后台运行** + ```bash + # Windows + start /b python fancontroller.py + + # Linux + nohup python3 fancontroller.py & + ``` + + ## 模式二:单次执行模式(推荐用于外部调度) + + 执行一次温度检测和风扇调整后退出,适合被 cron、systemd timer 等外部调度工具定时调用。 + + **直接执行** + ```bash + # Windows + python fancontroller_once.py + + # Linux + python3 fancontroller_once.py + ``` + + **使用 cron 定时执行(Linux)** + ```bash + # 编辑 crontab + crontab -e + + # 每 10 分钟执行一次 + */10 * * * * /usr/bin/python3 /path/to/python-ipmitool/fancontroller_once.py + ``` + + **使用 Windows 任务计划程序** + ```powershell + # 创建每 10 分钟执行一次的任务 + schtasks /create /tn "IPMI Fan Controller" /tr "python C:\path\to\python-ipmitool\fancontroller_once.py" /sc minute /mo 10 + ``` + +### 邮件告警功能(可选) + +项目支持在风扇调节失败时自动发送告警邮件,**默认关闭**。 + +#### 启用步骤 + +1. **配置邮箱信息** + + 编辑 `fan_settings.yaml`,设置 `alert.enabled: true` 并填写邮箱配置: + + ```yaml + alert: + enabled: true # 启用邮件告警 + fan_speed_threshold: 10000 # 风扇转速阈值(RPM) + max_failed_attempts: 3 # 连续失败3次后发送邮件 + email: + smtp_server: "smtp.gmail.com" + smtp_port: 587 + use_tls: true + sender_email: "your_email@gmail.com" + sender_password: "your_app_password" # Gmail 使用应用专用密码 + recipient_emails: + - "admin@example.com" + - "alert@example.com" # 支持多个收件人 + ``` + +2. **Gmail 邮箱配置** + + - 开启两步验证 + - 生成应用专用密码: https://myaccount.google.com/apppasswords + - 使用应用专用密码替代 Gmail 密码 + +3. **其他邮箱服务器** + + | 邮箱服务 | SMTP 服务器 | 端口 | TLS | + |---------|------------|------|-----| + | Gmail | smtp.gmail.com | 587 | true | + | QQ 邮箱 | smtp.qq.com | 587 | true | + | 163 邮箱 | smtp.163.com | 465 | false (使用SSL) | + | Outlook | smtp-mail.outlook.com | 587 | true | + +#### 告警触发条件 + +- 风扇转速超过配置的阈值(默认 10000 RPM) +- 连续检测失败达到配置的次数(默认 3 次) +- 为避免邮件轰炸,同一服务器告警邮件间隔至少 1 小时 + +#### 告警邮件内容 + +邮件包含: +- 服务器 IP 地址 +- 当前所有 CPU 温度 +- 当前所有风扇转速 +- 连续失败次数 +- 可能的故障原因和建议操作 ### 设置为 systemd 服务 (Linux 推荐) diff --git a/README_EN.md b/README_EN.md index cd8fb2d..ce884eb 100644 --- a/README_EN.md +++ b/README_EN.md @@ -34,60 +34,110 @@ The following models have been tested and are confirmed to work. More models are cd python-ipmitool ``` -3. **Copy the template file** `fan_settings.json.template` to `fan_settings.json`. +3. **Install dependencies** -4. **Edit the newly created `fan_settings.json`** file. The meaning of each field is as follows. You need to configure the IP addresses and fan speeds yourself. + ``` + pip install -r requirements.txt + ``` + +4. **Copy the template file** `fan_settings.yaml.template` to `fan_settings.yaml`. + + ```bash + # Linux/Mac + cp fan_settings.yaml.template fan_settings.yaml + + # Windows + copy fan_settings.yaml.template fan_settings.yaml + ``` + +5. **Edit the newly created `fan_settings.yaml`** file. The meaning of each field is as follows. You need to configure the IP addresses and fan speeds yourself. > Note: Only IP addresses are supported, not domain names. - > - > Note: When adding or removing server configurations, be careful not to leave a trailing comma `}` inside the last `}` of a list, as this will cause a `json.decoder.JSONDecodeError`. - - ```json - { - "auto": true, // true for automatic fan control, false for manual. - "interval": 60, // The interval in seconds for checking temperature and adjusting fan speed. - "log_backup_count": 30, // Number of days to retain log files. - "windows_ipmi_tool_path": ".\\ipmitool\\ipmitool.exe", // Path to the ipmitool executable on Windows. - "servers": [ // List of servers to manage. - { - "type": "dell730", // Server type. - "ip": "192.168.71.90", // Server IP address. Set to "local" if the script is running on the target machine. - "user": "root", // IPMI username. - "password": "123123", // IPMI password. - "temperature_ranges": [ // List of temperature ranges and corresponding fan speeds. - { - "min_temp": 0, // Minimum temperature of the range (inclusive). - "max_temp": 60, // Maximum temperature of the range (inclusive). - "fan_speeds": [20,20,20,20,20,20] // List of fan speeds in percent for this range. - }, - { - "min_temp": 61, - "max_temp": 80, - "fan_speeds": [25,25,25,25,25,25] - } - ] - } - ] - } - ``` - -5. **Run the Project** - - For long-term operation, it is recommended to run the script as a background process. - - 1. **On Windows** - - Use the `start /b` command to run the script in the background: - ``` - start /b python fancontroller.py - ``` - - 2. **On Linux** - - Use `nohup` and `&` to run the script in the background and ensure it keeps running after you close the terminal: - ``` -nohup python3 fancontroller.py & - ``` + + ```yaml + # IPMI Fan Controller Configuration + + # true for automatic fan control, false for manual + auto: true + + # The interval in seconds for checking temperature and adjusting fan speed + interval: 60 + + # Number of days to retain log files + log_backup_count: 30 + + # Path to the ipmitool executable on Windows + windows_ipmi_tool_path: ".\\ipmitool\\ipmitool.exe" + + # List of servers to manage + servers: + - type: dell730 # Server type + ip: "192.168.71.90" # Server IP address. Set to "local" if running on the target machine + user: root # IPMI username + password: "123123" # IPMI password + temperature_ranges: # List of temperature ranges and corresponding fan speeds + - min_temp: 0 # Minimum temperature of the range (inclusive) + max_temp: 60 # Maximum temperature of the range (inclusive) + fan_speeds: [20, 20, 20, 20, 20, 20] # List of fan speeds in percent + - min_temp: 61 + max_temp: 80 + fan_speeds: [25, 25, 25, 25, 25, 25] + ``` + + +6. **Run the Project** + + The project offers two execution modes: + + ## Mode 1: Loop Control Mode (Recommended for Long-term Operation) + + The program continuously monitors temperature and adjusts fan speeds. Suitable for running as a background service. + + **Foreground Execution (for debugging)** + ```bash + # Windows + python fancontroller.py + + # Linux + python3 fancontroller.py + ``` + + **Background Execution** + ```bash + # Windows + start /b python fancontroller.py + + # Linux + nohup python3 fancontroller.py & + ``` + + ## Mode 2: One-Shot Execution Mode (Recommended for External Scheduling) + + Executes once and exits after temperature detection and fan adjustment. Suitable for being called by external scheduling tools like cron, systemd timer, etc. + + **Direct Execution** + ```bash + # Windows + python fancontroller_once.py + + # Linux + python3 fancontroller_once.py + ``` + + **Using cron for Scheduled Execution (Linux)** + ```bash + # Edit crontab + crontab -e + + # Execute every 10 minutes + */10 * * * * /usr/bin/python3 /path/to/python-ipmitool/fancontroller_once.py + ``` + + **Using Windows Task Scheduler** + ```powershell + # Create a task that runs every 10 minutes + schtasks /create /tn "IPMI Fan Controller" /tr "python C:\path\to\python-ipmitool\fancontroller_once.py" /sc minute /mo 10 + ``` ### Setup as a systemd Service (Linux Recommended) diff --git a/fanController/base_controller.py b/fanController/base_controller.py index c04b47d..c62edae 100644 --- a/fanController/base_controller.py +++ b/fanController/base_controller.py @@ -5,7 +5,7 @@ from datetime import datetime class IPMIFanController: - def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True): + def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True, alert_config=None): """ 初始化 IPMI 风扇控制器。 @@ -15,6 +15,7 @@ def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True) windows_ipmi_tool_path (str): Windows 平台上 IPMI 工具的路径。 logger (logging.Logger): 配置好的日志记录器实例。 auto (bool): 是否自动模式,True为自动模式,False为手动模式。 + alert_config (dict, optional): 告警配置字典。 """ self.platform_system = platform.system() if self.platform_system == 'Windows': @@ -29,6 +30,32 @@ def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True) self.password = self.servers['password'] self.auto = auto + # 统计信息 + self.start_time = None + self.adjustment_count = 0 + self.last_cpu_temp = None + self.last_check_time = None + + # 告警配置 + self.alert_config = alert_config or {} + self.alert_enabled = self.alert_config.get('enabled', False) + self.fan_speed_threshold = self.alert_config.get('fan_speed_threshold', 10000) + self.max_failed_attempts = self.alert_config.get('max_failed_attempts', 3) + self.failed_attempts = 0 + self.last_alert_time = None + self.email_notifier = None + + # 初始化邮件通知器(如果启用) + if self.alert_enabled: + try: + from utils.email_notifier import EmailNotifier + email_config = self.alert_config.get('email', {}) + self.email_notifier = EmailNotifier(email_config, logger) + self.logger.info(f"服务器 {self.ip}: 邮件告警功能已启用 (阈值: {self.fan_speed_threshold} RPM, 失败次数: {self.max_failed_attempts})") + except Exception as e: + self.logger.error(f"初始化邮件通知器失败: {str(e)}") + self.alert_enabled = False + def send_command(self, cmd_in): """ 发送命令到系统 Shell。 @@ -96,61 +123,194 @@ def get_fan_rotational_speed(self): """ raise NotImplementedError("Method get_cpu_temperature must be implemented by subclasses") - def process_server(self): + def adjust_fans_once(self, prev_temp_ranges=None, prev_fan_speeds=None): """ - 处理服务器,监测 CPU 温度并相应调整风扇转速。 + 单次检测温度并调整风扇转速(不循环)。 + + Args: + prev_temp_ranges (tuple, optional): 之前的温度范围。 + prev_fan_speeds (list, optional): 之前的风扇转速。 + + Returns: + dict: 包含当前状态信息的字典 { + 'temp_ranges': (min_temp, max_temp) or None, + 'fan_speeds': [速度列表] or None, + 'cpu_temp': 当前最高CPU温度, + 'max_fan_speed': 当前最高风扇转速 + } """ - # 设置 IPMI 为手动模式 - self.set_ipmi_manual_mode() + check_start_time = time.time() + cpu_temps = self.get_cpu_temperature() + current_fan_speeds = self.get_fan_rotational_speed() - prev_temp_ranges = None - prev_fan_speeds = None + result = { + 'temp_ranges': None, + 'fan_speeds': None, + 'cpu_temp': None, + 'max_fan_speed': None + } - self.monitor_and_adjust_fans(prev_temp_ranges, prev_fan_speeds) + if cpu_temps: + max_temp_value = max(cpu_temps) + min_temp_value = min(cpu_temps) + avg_temp_value = sum(cpu_temps) // len(cpu_temps) + max_fan_speed = max(current_fan_speeds) + min_fan_speed = min(current_fan_speeds) + result['cpu_temp'] = max_temp_value + result['max_fan_speed'] = max_fan_speed - def monitor_and_adjust_fans(self, prev_temp_ranges, prev_fan_speeds): - """ - 监控并调整风扇转速的方法。 + # 计算温度变化 + temp_change_str = "" + if self.last_cpu_temp is not None: + temp_delta = max_temp_value - self.last_cpu_temp + if temp_delta > 0: + temp_change_str = f" | 变化: +{temp_delta}°C ↑" + if temp_delta >= 10: + temp_change_str += " [警告: 温度快速上升!]" + elif temp_delta < 0: + temp_change_str = f" | 变化: {temp_delta}°C ↓" + else: + temp_change_str = f" | 变化: 持平 →" + + # 计算实际检测间隔 + interval_str = "" + if self.last_check_time is not None: + actual_interval = time.time() - self.last_check_time + interval_str = f" | 检测间隔: {actual_interval:.1f}秒" + + # 详细日志:显示所有 CPU 温度和风扇转速 + cpu_temps_str = ', '.join([f"{temp}°C" for temp in cpu_temps]) + fan_speeds_str = ', '.join([f"{speed} RPM" for speed in current_fan_speeds]) + + self.logger.info("=" * 80) + self.logger.info(f"服务器 {self.ip} 状态检测:") + self.logger.info(f" CPU 温度: [{cpu_temps_str}]") + self.logger.info(f" └─ 最低: {min_temp_value}°C | 平均: {avg_temp_value}°C | 最高: {max_temp_value}°C{temp_change_str}") + self.logger.info(f" 风扇转速: [{fan_speeds_str}]") + self.logger.info(f" └─ 最低: {min_fan_speed} RPM | 最高: {max_fan_speed} RPM") + + # 温度过高警告 + temp_threshold_warning = 75 # 可配置的警告阈值 + temp_threshold_critical = 85 # 可配置的严重阈值 + if max_temp_value >= temp_threshold_critical: + self.logger.warning(f" ⚠️ 严重警告: CPU 温度过高 ({max_temp_value}°C >= {temp_threshold_critical}°C)!") + elif max_temp_value >= temp_threshold_warning: + self.logger.warning(f" ⚠️ 警告: CPU 温度较高 ({max_temp_value}°C >= {temp_threshold_warning}°C)") + + # 风扇转速异常检测(告警前的检查) + if max_fan_speed >= self.fan_speed_threshold: + self.logger.warning(f" ⚠️ 风扇转速异常: {max_fan_speed} RPM (阈值: {self.fan_speed_threshold} RPM)") + + # 如果启用告警,检查是否需要发送告警邮件 + if self.alert_enabled: + self.failed_attempts += 1 + self.logger.warning(f" 风扇调节失败计数: {self.failed_attempts}/{self.max_failed_attempts}") + + # 连续失败次数达到阈值,发送告警邮件 + if self.failed_attempts >= self.max_failed_attempts: + # 避免频繁发送邮件,至少间隔1小时 + current_time = time.time() + should_send = True + if self.last_alert_time is not None: + time_since_last_alert = current_time - self.last_alert_time + if time_since_last_alert < 3600: # 1小时 + should_send = False + self.logger.info(f" 距离上次告警仅 {time_since_last_alert/60:.1f} 分钟,跳过邮件发送") + + if should_send and self.email_notifier: + self.logger.warning(f" ⚠️⚠️⚠️ 连续 {self.failed_attempts} 次调节失败,发送告警邮件!") + subject = f"🚨 IPMI 风扇控制器告警 - 服务器 {self.ip}" + success = self.email_notifier.send_alert( + subject=subject, + server_ip=self.ip, + cpu_temps=cpu_temps, + fan_speeds=current_fan_speeds, + failed_attempts=self.failed_attempts, + threshold=self.fan_speed_threshold + ) + if success: + self.last_alert_time = current_time + # 发送成功后重置计数器,避免重复告警 + self.failed_attempts = 0 + else: + # 风扇转速正常,重置失败计数 + if self.failed_attempts > 0: + self.logger.info(f" 风扇转速已恢复正常 ({max_fan_speed} RPM < {self.fan_speed_threshold} RPM),重置失败计数") + self.failed_attempts = 0 + + for temp_range in self.servers['temperature_ranges']: + min_temp = temp_range['min_temp'] + max_temp = temp_range['max_temp'] + fan_speeds = temp_range['fan_speeds'] + + if min_temp <= max_temp_value <= max_temp: + if (prev_temp_ranges == (min_temp, max_temp)) and (prev_fan_speeds == fan_speeds) and max_fan_speed < 15000: + self.logger.info(f" 动作: 温度在范围 [{min_temp}-{max_temp}°C] 内,风扇转速保持不变") + result['temp_ranges'] = (min_temp, max_temp) + result['fan_speeds'] = fan_speeds + else: + fan_speeds_percent_str = ', '.join([f"{speed}%" for speed in fan_speeds]) + self.logger.info(f" 动作: 温度 {max_temp_value}°C 在范围 [{min_temp}-{max_temp}°C],设置风扇转速为 [{fan_speeds_percent_str}]") - Args: - prev_temp_ranges (tuple): 之前的温度范围。 - prev_fan_speeds (list): 之前的风扇转速。 - """ - while True: - cpu_temps = self.get_cpu_temperature() - current_fan_speeds = self.get_fan_rotational_speed() - current_time = datetime.now().strftime("%Y-%m-%d %H:%M:%S") - if cpu_temps: - avg_temp = max(cpu_temps) - max_speed = max(current_fan_speeds) - log_message = f"服务器 {self.ip}:CPU 平均温度:{avg_temp}°C, 风扇最大转速{max_speed}" - self.logger.info(log_message) # 使用 logging 记录日志 - - for temp_range in self.servers['temperature_ranges']: - min_temp = temp_range['min_temp'] - max_temp = temp_range['max_temp'] - fan_speeds = temp_range['fan_speeds'] - - if min_temp <= avg_temp < max_temp: - if (prev_temp_ranges == (min_temp, max_temp)) and (prev_fan_speeds == fan_speeds) and max_speed < 15000: - log_message = "温度在之前的范围内,跳过设置" - self.logger.info(log_message) # 使用 logging 记录日志 - break - - log_message = f"设置风扇转速为 {fan_speeds}" - self.logger.info(log_message) # 使用 logging 记录日志 for fan_index, speed in enumerate(fan_speeds): self.set_fan_speed(fan_index, speed) time.sleep(1) - prev_temp_ranges = (min_temp, max_temp) - prev_fan_speeds = fan_speeds - break - else: - log_message = f"服务器 {self.ip}:没有 CPU 温度数据可用。" - self.logger.info(log_message) # 使用 logging 记录日志 + self.adjustment_count += 1 + self.logger.info(f" 风扇调整完成 (总调整次数: {self.adjustment_count})") + + result['temp_ranges'] = (min_temp, max_temp) + result['fan_speeds'] = fan_speeds + break + + # 更新状态 + self.last_cpu_temp = max_temp_value + self.last_check_time = time.time() - if not self.auto: - break + # 性能统计 + check_duration = time.time() - check_start_time + self.logger.info(f" 性能: 检测耗时 {check_duration:.2f}秒{interval_str}") + + # 运行时统计(仅循环模式) + if self.start_time is not None: + uptime = time.time() - self.start_time + uptime_hours = uptime / 3600 + if uptime_hours >= 1: + self.logger.info(f" 统计: 运行时长 {uptime_hours:.1f}小时 | 调整次数 {self.adjustment_count}") + + else: + self.logger.error(f"服务器 {self.ip}: 没有 CPU 温度数据可用!") + + return result + + def process_server_loop(self): + """ + 循环模式:持续监测 CPU 温度并相应调整风扇转速。 + """ + # 设置 IPMI 为手动模式 + self.set_ipmi_manual_mode() + + # 初始化统计信息 + self.start_time = time.time() + self.logger.info(f"服务器 {self.ip}: 循环控制模式已启动,检测间隔 {self.interval} 秒") + + prev_temp_ranges = None + prev_fan_speeds = None + + while True: + result = self.adjust_fans_once(prev_temp_ranges, prev_fan_speeds) + prev_temp_ranges = result['temp_ranges'] + prev_fan_speeds = result['fan_speeds'] time.sleep(self.interval) + + def process_server_once(self): + """ + 单次执行模式:执行一次温度检测和风扇调整后退出。 + 适合被外部调度工具(cron、systemd timer 等)调用。 + """ + # 设置 IPMI 为手动模式 + self.set_ipmi_manual_mode() + + # 执行一次调整 + self.adjust_fans_once() diff --git a/fanController/dell730_controller.py b/fanController/dell730_controller.py index 95d72c8..2a93a77 100644 --- a/fanController/dell730_controller.py +++ b/fanController/dell730_controller.py @@ -1,4 +1,5 @@ import re +import time import logging from .base_controller import IPMIFanController # 导入基础控制器类 @@ -63,11 +64,48 @@ def get_fan_rotational_speed(self): return rpm_values - def start_fan_control(self): - """启动风扇控制的方法。 - 使用self.auto来判断是否自动模式。 + def _initialize_dell730(self): + """ + Dell 730 服务器初始化流程。 + """ + # Dell 730 特定初始化:禁用第三方 PCIe 卡的散热响应策略 + self.disable_third_party_pcie_thermal_response() + # 设置 IPMI 为手动模式 + self.set_ipmi_manual_mode() + + def process_server_loop(self): + """ + 循环模式:持续监测 CPU 温度并相应调整风扇转速。 + 重写父类方法以添加 Dell 特定的初始化步骤。 + """ + self._initialize_dell730() + + prev_temp_ranges = None + prev_fan_speeds = None + + while True: + result = self.adjust_fans_once(prev_temp_ranges, prev_fan_speeds) + prev_temp_ranges = result['temp_ranges'] + prev_fan_speeds = result['fan_speeds'] + + time.sleep(self.interval) + + def process_server_once(self): + """ + 单次执行模式:执行一次温度检测和风扇调整后退出。 + 重写父类方法以添加 Dell 特定的初始化步骤。 """ - self.process_server() + self._initialize_dell730() + # 执行一次调整 + self.adjust_fans_once() + + def start_fan_control(self): + """启动风扇控制的方法(循环模式)。""" + self.process_server_loop() + + def run_once(self): + """执行一次风扇控制(单次模式)。""" + self.process_server_once() def set_ipmi_manual_mode(self): """ @@ -77,7 +115,7 @@ def set_ipmi_manual_mode(self): str: IPMI 命令的输出。 """ base_cmd = self._get_base_command() - command = f'{{base_cmd}} raw 0x30 0x30 0x01 0x00' + command = f'{base_cmd} raw 0x30 0x30 0x01 0x00' return self.ipmi_command(command.strip()) def set_ipmi_auto_mode(self): @@ -88,5 +126,25 @@ def set_ipmi_auto_mode(self): str: IPMI 命令的输出。 """ base_cmd = self._get_base_command() - command = f'{{base_cmd}} raw 0x30 0x30 0x01 0x01' + command = f'{base_cmd} raw 0x30 0x30 0x01 0x01' return self.ipmi_command(command.strip()) + + def disable_third_party_pcie_thermal_response(self): + """ + 禁用第三方 PCIe 卡的散热响应策略。 + 这是 Dell 服务器风扇控制的必要初始化步骤。 + + Returns: + str: IPMI 命令的输出。成功时应返回 "16 05 00 00 00"。 + """ + base_cmd = self._get_base_command() + command = f'{base_cmd} raw 0x30 0xce 0x00 0x16 0x05 0x00 0x00 0x00 0x05 0x00 0x01 0x00 0x00' + output = self.ipmi_command(command.strip()) + + # 记录命令执行结果 + if '16 05 00 00 00' in output.replace(' ', '').replace('\n', ''): + self.logger.info(f"服务器 {self.ip}: 成功禁用第三方 PCIe 卡散热响应策略") + else: + self.logger.warning(f"服务器 {self.ip}: 禁用第三方 PCIe 卡散热响应策略可能失败,返回: {output.strip()}") + + return output diff --git a/fan_settings.json.template b/fan_settings.json.template deleted file mode 100644 index 3286fb0..0000000 --- a/fan_settings.json.template +++ /dev/null @@ -1,62 +0,0 @@ -{ - "auto": true, - "interval": 60, - "log_backup_count": 30, - "windows_ipmi_tool_path": ".\\ipmitool\\ipmitool.exe", - "servers": [ - { - "type": "dell730", - "ip": "192.168.71.90", - "user": "root", - "password": "123123", - "temperature_ranges": [ - { - "min_temp": 0, - "max_temp": 60, - "fan_speeds": [20,20,20,20,20,20] - }, - { - "min_temp": 61, - "max_temp": 80, - "fan_speeds": [25,25,25,25,25,25] - } - ] - }, - { - "type": "dell730", - "ip": "192.168.71.91", - "user": "root", - "password": "123123", - "temperature_ranges": [ - { - "min_temp": 0, - "max_temp": 60, - "fan_speeds": [20,20,20,20,20,20] - }, - { - "min_temp": 61, - "max_temp": 80, - "fan_speeds": [25,25,25,25,25,25] - } - ] - }, - { - "type": "dell730", - "ip": "192.168.71.92", - "user": "root", - "password": "123123", - "temperature_ranges": [ - { - "min_temp": 0, - "max_temp": 60, - "fan_speeds": [20,20,20,20,20,20] - }, - { - "min_temp": 61, - "max_temp": 80, - "fan_speeds": [25,25,25,25,25,25] - } - ] - } - ] -} \ No newline at end of file diff --git a/fan_settings.yaml.template b/fan_settings.yaml.template new file mode 100644 index 0000000..8b4d92e --- /dev/null +++ b/fan_settings.yaml.template @@ -0,0 +1,84 @@ +# IPMI 风扇控制器配置文件 + +# 是否自动控制风扇转速,true 为自动控制,false 为手动控制 +auto: true + +# 控制风扇转速的时间间隔,单位为秒 +interval: 600 + +# 日志文件保留天数 +log_backup_count: 30 + +# Windows 系统下 ipmitool 工具的路径 +windows_ipmi_tool_path: ".\\ipmitool\\ipmitool.exe" + +# 告警配置 +alert: + # 是否启用邮件告警功能,默认 false 关闭 + enabled: false + + # 风扇转速异常阈值(RPM),超过此值视为调节失败 + fan_speed_threshold: 10000 + + # 连续调节失败次数阈值,达到此次数后发送告警邮件 + max_failed_attempts: 3 + + # 邮件服务器配置 + email: + # SMTP 服务器地址 + smtp_server: "smtp.163.com" + + # SMTP 端口(通常 465 为 SSL,587 为 TLS) + smtp_port: 587 + + # 是否使用 TLS 加密 + use_tls: true + + # 发件人邮箱 + sender_email: "your_email@gmail.com" + + # 发件人邮箱密码或应用专用密码 + sender_password: "your_app_password" + + # 收件人邮箱列表(支持多个收件人) + recipient_emails: + - "admin@example.com" + - "alert@example.com" + +# 服务器列表 +servers: + - type: dell730 # 服务器类型 + ip: "192.168.71.90" # 服务器 IP 地址。如果脚本与服务器在同一台机器运行,可设置为 "local" 以直接执行本地命令 + user: root # IPMI 用户名 + password: "123123" # IPMI 密码 + temperature_ranges: # 温度范围与对应的风扇转速 + - min_temp: 0 # 区间最低温度(包括) + max_temp: 60 # 区间最高温度(包括) + fan_speeds: [20, 20, 20, 20, 20, 20] # 对应风扇转速的列表,单位为百分比 + - min_temp: 61 + max_temp: 80 + fan_speeds: [25, 25, 25, 25, 25, 25] + + - type: dell730 + ip: "192.168.71.91" + user: root + password: "123123" + temperature_ranges: + - min_temp: 0 + max_temp: 60 + fan_speeds: [20, 20, 20, 20, 20, 20] + - min_temp: 61 + max_temp: 80 + fan_speeds: [25, 25, 25, 25, 25, 25] + + - type: dell730 + ip: "192.168.71.92" + user: root + password: "123123" + temperature_ranges: + - min_temp: 0 + max_temp: 60 + fan_speeds: [20, 20, 20, 20, 20, 20] + - min_temp: 61 + max_temp: 80 + fan_speeds: [25, 25, 25, 25, 25, 25] diff --git a/fancontroller.py b/fancontroller.py index d7e21d7..e6bb6d9 100644 --- a/fancontroller.py +++ b/fancontroller.py @@ -1,8 +1,25 @@ -import json +""" +IPMI 风扇控制器 - 循环执行模式 + +此脚本持续运行,定期监控温度并调整风扇转速。 +适合作为后台服务或 systemd 服务运行。 + +使用场景: +- 作为后台进程持续运行 +- 通过 systemd 服务管理 +- 需要实时响应温度变化的场景 + +优势: +- 实时监控,响应及时 +- 保持上下文状态,避免重复初始化 +- 适合长期运行的服务器环境 +""" + import os import threading import logging from logging.handlers import TimedRotatingFileHandler +import yaml from fanController.dell730_controller import Dell730FanController @@ -15,17 +32,16 @@ def main(): log_file_path = os.path.join(log_directory, 'fancontroller.log') # --- 读取配置 --- - config_file_path = os.path.join(current_directory, 'fan_settings.json') + config_file_path = os.path.join(current_directory, 'fan_settings.yaml') try: with open(config_file_path, 'r', encoding='utf-8') as file: - data = json.load(file) + data = yaml.safe_load(file) except FileNotFoundError: - # 在日志系统完全建立前,只能用print - print(f"错误:配置文件 'fan_settings.json' 未找到。") + print(f"错误:配置文件 'fan_settings.yaml' 未找到。") input("按任意键退出程序:") return - except json.JSONDecodeError: - print("错误:文件内容不是有效的JSON格式,请检查配置后重新打开。") + except yaml.YAMLError as e: + print(f"错误:配置文件格式错误,请检查配置后重新打开。\n详细信息:{e}") input("按任意键退出程序:") return @@ -57,18 +73,24 @@ def main(): logger.addHandler(file_handler) logger.addHandler(stream_handler) - # --- 启动控制器 --- - logger.info("程序启动") + # --- 启动控制器(循环模式)--- + logger.info("循环控制模式启动") servers = data['servers'] windows_ipmi_tool_path = data['windows_ipmi_tool_path'] interval = data['interval'] - auto = data.get('auto', False) + alert_config = data.get('alert', {}) # 获取告警配置 threads = [] for server in servers: if server['type'] == 'dell730': - fan_controller = Dell730FanController(servers=server, interval=interval, - windows_ipmi_tool_path=windows_ipmi_tool_path, logger=logger, auto=auto) + fan_controller = Dell730FanController( + servers=server, + interval=interval, + windows_ipmi_tool_path=windows_ipmi_tool_path, + logger=logger, + auto=True, # 循环模式 + alert_config=alert_config + ) thread = threading.Thread(target=fan_controller.start_fan_control, name=f"Thread-{server['ip']}") thread.start() threads.append(thread) diff --git a/fancontroller_once.py b/fancontroller_once.py new file mode 100644 index 0000000..c112ede --- /dev/null +++ b/fancontroller_once.py @@ -0,0 +1,97 @@ +""" +IPMI 风扇控制器 - 单次执行模式 + +此脚本执行一次温度检测和风扇调整后退出。 +适合被外部调度工具(cron、systemd timer、Windows 任务计划程序等)定时调用。 + +使用场景: +- 由 cron 或 systemd timer 每隔 N 分钟调用一次 +- 由外部监控系统触发执行 +- 集成到其他自动化工作流中 + +优势: +- 更灵活的调度控制 +- 便于集成到现有的任务调度系统 +- 执行失败不会影响后续调度 +""" + +import os +import logging +from logging.handlers import TimedRotatingFileHandler +import yaml + +from fanController.dell730_controller import Dell730FanController + + +def main(): + # --- 路径设置 --- + current_directory = os.path.dirname(os.path.abspath(__file__)) + log_directory = os.path.join(current_directory, 'logs') + os.makedirs(log_directory, exist_ok=True) + log_file_path = os.path.join(log_directory, 'fancontroller.log') + + # --- 读取配置 --- + config_file_path = os.path.join(current_directory, 'fan_settings.yaml') + try: + with open(config_file_path, 'r', encoding='utf-8') as file: + data = yaml.safe_load(file) + except FileNotFoundError: + print(f"错误:配置文件 'fan_settings.yaml' 未找到。") + return + except yaml.YAMLError as e: + print(f"错误:配置文件格式错误,请检查配置后重新打开。\n详细信息:{e}") + return + + # --- 日志配置 --- + log_backup_count = data.get('log_backup_count', 30) + logger = logging.getLogger('FanController') + logger.setLevel(logging.INFO) + + # 文件处理器 (按天轮转) + file_handler = TimedRotatingFileHandler( + log_file_path, + when='midnight', + interval=1, + backupCount=log_backup_count, + encoding='utf-8' + ) + file_handler.setLevel(logging.INFO) + + # 控制台处理器 + stream_handler = logging.StreamHandler() + stream_handler.setLevel(logging.INFO) + + # 日志格式 + formatter = logging.Formatter('%(asctime)s - %(message)s') + file_handler.setFormatter(formatter) + stream_handler.setFormatter(formatter) + + # 添加处理器到logger + logger.addHandler(file_handler) + logger.addHandler(stream_handler) + + # --- 执行单次控制 --- + logger.info("单次执行模式启动") + servers = data['servers'] + windows_ipmi_tool_path = data['windows_ipmi_tool_path'] + interval = data.get('interval', 60) # 单次模式不使用 interval,但保留参数兼容性 + alert_config = data.get('alert', {}) # 获取告警配置 + + for server in servers: + if server['type'] == 'dell730': + fan_controller = Dell730FanController( + servers=server, + interval=interval, + windows_ipmi_tool_path=windows_ipmi_tool_path, + logger=logger, + auto=False, # 单次模式不需要 auto 参数 + alert_config=alert_config + ) + # 执行一次风扇控制 + fan_controller.run_once() + + logger.info("单次执行完成") + + +if __name__ == '__main__': + main() diff --git a/requirements.txt b/requirements.txt index e69de29..043876c 100644 --- a/requirements.txt +++ b/requirements.txt @@ -0,0 +1 @@ +PyYAML>=6.0 \ No newline at end of file diff --git a/utils/__init__.py b/utils/__init__.py new file mode 100644 index 0000000..09e358a --- /dev/null +++ b/utils/__init__.py @@ -0,0 +1,7 @@ +""" +工具模块 +""" + +from .email_notifier import EmailNotifier + +__all__ = ['EmailNotifier'] diff --git a/utils/email_notifier.py b/utils/email_notifier.py new file mode 100644 index 0000000..9f4f5ee --- /dev/null +++ b/utils/email_notifier.py @@ -0,0 +1,156 @@ +""" +邮件通知工具模块 + +用于发送告警邮件通知 +""" + +import smtplib +from email.mime.text import MIMEText +from email.mime.multipart import MIMEMultipart +from datetime import datetime +import logging + + +class EmailNotifier: + """邮件通知类""" + + def __init__(self, config, logger=None): + """ + 初始化邮件通知器 + + Args: + config (dict): 邮件配置字典,包含 smtp_server, smtp_port, use_tls, + sender_email, sender_password, recipient_emails + logger (logging.Logger, optional): 日志记录器 + """ + self.smtp_server = config.get('smtp_server') + self.smtp_port = config.get('smtp_port', 587) + self.use_tls = config.get('use_tls', True) + self.sender_email = config.get('sender_email') + self.sender_password = config.get('sender_password') + self.recipient_emails = config.get('recipient_emails', []) + self.logger = logger or logging.getLogger(__name__) + + # 验证配置 + if not all([self.smtp_server, self.sender_email, self.sender_password, self.recipient_emails]): + self.logger.warning("邮件配置不完整,邮件通知功能将无法使用") + + def send_alert(self, subject, server_ip, cpu_temps, fan_speeds, failed_attempts, threshold): + """ + 发送告警邮件 + + Args: + subject (str): 邮件主题 + server_ip (str): 服务器 IP + cpu_temps (list): CPU 温度列表 + fan_speeds (list): 风扇转速列表 + failed_attempts (int): 连续失败次数 + threshold (int): 风扇转速阈值 + + Returns: + bool: 发送成功返回 True,失败返回 False + """ + try: + # 构建邮件内容 + message = MIMEMultipart() + message['From'] = self.sender_email + message['To'] = ', '.join(self.recipient_emails) + message['Subject'] = subject + + # 邮件正文 + max_temp = max(cpu_temps) if cpu_temps else 0 + max_fan_speed = max(fan_speeds) if fan_speeds else 0 + cpu_temps_str = ', '.join([f"{temp}°C" for temp in cpu_temps]) + fan_speeds_str = ', '.join([f"{speed} RPM" for speed in fan_speeds]) + + body = f""" + + + + + + +
+

⚠️ IPMI 风扇控制器告警

+

告警时间: {datetime.now().strftime("%Y-%m-%d %H:%M:%S")}

+

服务器 IP: {server_ip}

+

告警原因: 风扇转速调节失败,连续 {failed_attempts} 次调节后风扇转速仍超过阈值 {threshold} RPM

+
+ +

当前状态

+ + + + + + + + + + + + + + + + + + + + + +
项目详细信息
CPU 温度{cpu_temps_str}
最高温度: {max_temp}°C
风扇转速{fan_speeds_str}
最高转速: {max_fan_speed} RPM
失败次数{failed_attempts} 次
转速阈值{threshold} RPM
+ +

可能原因

+ + +

建议操作

+
    +
  1. 检查服务器 IPMI 连接是否正常
  2. +
  3. 检查服务器温度是否异常过高
  4. +
  5. 检查日志文件获取详细错误信息
  6. +
  7. 考虑手动介入,检查服务器硬件状态
  8. +
  9. 如持续告警,建议停止自动控制,让服务器自动管理风扇
  10. +
+ +

+ 此邮件由 IPMI 风扇控制器自动发送,请勿直接回复。 +

+ + +""" + + message.attach(MIMEText(body, 'html')) + + # 发送邮件 + if self.use_tls: + server = smtplib.SMTP(self.smtp_server, self.smtp_port) + server.starttls() + else: + server = smtplib.SMTP_SSL(self.smtp_server, self.smtp_port) + + server.login(self.sender_email, self.sender_password) + server.send_message(message) + server.quit() + + self.logger.info(f"告警邮件已发送至: {', '.join(self.recipient_emails)}") + return True + + except Exception as e: + self.logger.error(f"发送告警邮件失败: {str(e)}") + return False From e2f9299f3f2e8c507073be458d9649b953a92a35 Mon Sep 17 00:00:00 2001 From: dongyu6 <92629383+dongyu6@users.noreply.github.com> Date: Tue, 17 Mar 2026 17:37:24 +0800 Subject: [PATCH 02/13] =?UTF-8?q?=E6=B5=AA=E6=BD=AEipmi=E6=8E=A7=E5=88=B6?= =?UTF-8?q?=E9=A3=8E=E6=89=87=E4=BB=A3=E7=A0=81?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- set_fan_speed.py | 277 +++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 277 insertions(+) create mode 100644 set_fan_speed.py diff --git a/set_fan_speed.py b/set_fan_speed.py new file mode 100644 index 0000000..3d5e1e3 --- /dev/null +++ b/set_fan_speed.py @@ -0,0 +1,277 @@ +#!/usr/bin/python +# -*- coding: UTF-8 -*- +# +# 运行此脚本前,请确保已安装所需库: +# pip install requests pycryptodome +# +import requests +import socket +import time +import base64 +from Crypto.Cipher import Blowfish +from Crypto.Util.Padding import pad +import urllib3 + +# 禁用 InsecureRequestWarning 警告 +urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning) + +# --- 配置信息 --- +dic = { + 'username': 'zlkj', + 'password': 'Zl123456.', + 'ip': '192.168.20.11', + 'speed': '1', # 需要调整的转速百分比 + 'fans': ["1", "3", "5", "7"] # 需要调整转速的风扇端口 (0-7) +} + +# --- 加密算法实现 --- + +def encry_str(s: str) -> str: + """ + 使用XOR 127和十六进制编码来混淆字符串,模拟JavaScript中的encryStr函数。 + """ + if not s: + return "" + return '-'.join([hex(ord(char) ^ 127)[2:] for char in s]) + +def encrypt_blowfish(text: str, key: str) -> str: + """ + 使用Blowfish ECB模式加密文本,然后进行Base64编码。 + """ + cipher = Blowfish.new(key.encode('utf-8'), Blowfish.MODE_ECB) + padded_text = pad(text.encode('utf-8'), Blowfish.block_size) + encrypted_text = cipher.encrypt(padded_text) + return base64.b64encode(encrypted_text).decode('utf-8') + +# --- 核心功能 --- + +def is_port_open(ip_address, port): + """检查指定IP的端口是否开放""" + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.settimeout(2) + result = sock.connect_ex((ip_address, port)) + sock.close() + return result == 0 + +def login(ip, username, password): + """ + 执行登录流程,返回一个包含认证cookie和CSRF令牌的requests.Session对象。 + """ + print("开始登录 bmc ------>") + base_url = f"https://{ip}" + session = requests.Session() + # 添加 'X-Requested-With' 头,模拟AJAX请求 + session.headers.update({ + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36', + 'Referer': f"{base_url}/main.html", + 'X-Requested-With': 'XMLHttpRequest' + }) + + # 1. 获取加密模式和登录标签 + try: + print("步骤 1/2: 获取加密模式...") + random_tag_resp = session.get(f"{base_url}/api/randomtag", timeout=10, verify=False) + random_tag_resp.raise_for_status() + login_data = random_tag_resp.json() + encrypt_ctrl = login_data.get('encrypt_ctrl', 0) + login_tag = login_data.get('random') + print(f"获取成功, 加密模式 (encrypt_ctrl): {encrypt_ctrl}") + except requests.RequestException as e: + print(f"获取加密模式失败: {e}") + return None + + # 2. 根据加密模式加密凭据并登录 + encrypt_username = username + encrypt_password = password + + if encrypt_ctrl == 1: + print("使用 XOR 加密...") + encrypt_username = encry_str(username) + encrypt_password = encry_str(password) + elif encrypt_ctrl == 2: + print("使用 Blowfish 加密...") + secret_key = "secret" + encrypt_username = encrypt_blowfish(username, secret_key) + encrypt_password = encrypt_blowfish(password, secret_key) + else: + print("使用明文...") + + login_payload = { + 'username': encrypt_username, + 'password': encrypt_password, + 'encrypt_flag': encrypt_ctrl, + 'login_tag': login_tag + } + + try: + print("步骤 2/2: 发送登录请求...") + login_resp = session.post(f"{base_url}/api/session", data=login_payload, timeout=10, verify=False) + login_resp.raise_for_status() + login_result = login_resp.json() + + if login_result.get('ok') == 0 and 'CSRFToken' in login_result: + csrf_token = login_result['CSRFToken'] + session.headers.update({'X-CSRFTOKEN': csrf_token}) + print(f"登录成功! CSRF Token已设置。") + print('——————————————————————————————————————————————————————————————') + return session + else: + error_msg = login_result.get('error_msg', '未知错误') + if 'CSRFToken' not in login_result: + error_msg = "登录响应中未找到 CSRFToken。" + print(f"登录失败: {error_msg}") + return None + + except requests.RequestException as e: + print(f"登录请求失败: {e}") + return None + +def get_fan_mode(session, ip): + """获取当前风扇控制模式""" + print("正在获取当前风扇模式...") + url = f"https://{ip}/api/settings/fans-mode" + try: + response = session.get(url, timeout=10, verify=False) + response.raise_for_status() + mode = response.json().get("control_mode", "未知") + print(f"获取成功, 当前风扇模式为: {mode}") + return mode + except requests.RequestException as e: + print(f"获取风扇模式失败: {e}") + return None + +def set_fan_mode(session, ip, mode): + """设置风扇控制模式 ('manual' 或 'auto')""" + print(f"开始设置风扇模式为: {mode}...") + url = f"https://{ip}/api/settings/fans-mode" + payload = {'control_mode': mode} + try: + response = session.post(url, json=payload, timeout=10, verify=False) + response.raise_for_status() + print(f"已成功发送设置请求,目标模式: {mode}") + print('——————————————————————————————————————————————————————————————') + return True + except requests.RequestException as e: + print(f"设置风扇模式失败: {e}") + return False + +def set_fan_speed(session, ip, fan_id, speed_percent): + """设置单个风扇的速度""" + print(f"开始调整风扇 {fan_id} 的速度为 {speed_percent}%") + url = f"https://{ip}/api/settings/fan/{fan_id}" + payload = {'duty': speed_percent} + try: + response = session.put(url, json=payload, timeout=10, verify=False) + response.raise_for_status() + print(f"调整风扇 {fan_id} 完成。") + return True + except requests.RequestException as e: + error_message = f"调整风扇 {fan_id} 失败: {e}" + if e.response is not None: + error_message += f"\n - Status Code: {e.response.status_code}" + error_message += f"\n - Response Body: {e.response.text}" + print(error_message) + return False + +def get_fan_info(session, ip): + """获取所有风扇的详细信息""" + print("正在获取所有风扇的当前状态...") + url = f"https://{ip}/api/status/fan_info" + try: + response = session.get(url, timeout=10, verify=False) + response.raise_for_status() + fan_info = response.json() + print("获取风扇状态成功。") + return fan_info.get("fans", []) + except (requests.RequestException, ValueError) as e: + print(f"获取风扇状态失败: {e}") + return None + +# --- 主程序 --- +if __name__ == "__main__": + print("感谢使用修改风扇速度脚本") + print(f'目标IP: {dic["ip"]}') + print(f'用户名: {dic["username"]}') + print(f'待调整速度: {dic["speed"]}%') + print('——————————————————————————————————————————————————————————————') + + # 检查端口连通性 + print("正在测试BMC控制台端口连通性...") + if not is_port_open(dic['ip'], 443): + print(f"无法连接到 {dic['ip']}:443 (HTTPS)。请检查网络连接或IP地址。") + exit() + print("端口测试通过。") + print('——————————————————————————————————————————————————————————————') + + # 登录 + auth_session = login(dic['ip'], dic['username'], dic['password']) + + if not auth_session: + print("无法完成登录,脚本退出。") + exit() + + # 获取当前模式 + current_mode = get_fan_mode(auth_session, dic['ip']) + print('——————————————————————————————————————————————————————————————') + + if current_mode == 'manual': + print("风扇已处于手动模式,无需再次设置。") + else: + # 设置为手动模式 + if not set_fan_mode(auth_session, dic['ip'], 'manual'): + print("无法设置风扇为手动模式,脚本退出。") + exit() + + # 再次获取模式以确认更改 + print("等待2秒后确认模式...") + time.sleep(2) + get_fan_mode(auth_session, dic['ip']) + print('——————————————————————————————————————————————————————————————') + + # 获取所有风扇的当前状态 + all_fans_info = get_fan_info(auth_session, dic['ip']) + if all_fans_info is None: + print("无法获取风扇信息,脚本退出。") + exit() + + # 将风扇列表转换为以ID为键的字典,方便快速查找 + # 注意:JSON中的id是数字,而配置中的是字符串 + fan_status_map = {str(fan['id']): fan for fan in all_fans_info} + + # 调整所有指定风扇的速度 + print('开始检查并调整所有指定风扇的风速...') + all_success = True + # 将配置中的速度转换为整数以便比较 + target_speed = int(dic['speed']) + + for fan_id_str in dic['fans']: + if fan_id_str not in fan_status_map: + print(f"警告: 在风扇信息中未找到ID为 {fan_id_str} 的风扇,跳过调整。") + continue + + current_fan = fan_status_map[fan_id_str] + current_speed = current_fan.get('speed_percent') + + if current_speed is None: + print(f"警告: 无法获取风扇 {fan_id_str} 的当前速度,跳过调整。") + continue + + print(f"检查风扇 {fan_id_str}: 当前速度 {current_speed}%, 目标速度 {target_speed}%") + + if current_speed == target_speed: + print(f"风扇 {fan_id_str} 的速度已是 {target_speed}%, 无需调整。") + else: + # 调用设置函数时,仍然使用配置中的原始字符串格式的速度值 + if not set_fan_speed(auth_session, dic['ip'], fan_id_str, dic['speed']): + all_success = False + print('---') # 为每个风扇的处理添加分隔符,使输出更清晰 + + print('——————————————————————————————————————————————————————————————') + + if all_success: + print('所有风扇速度调整任务已成功完成。') + else: + print('部分风扇速度调整失败,请检查上面的日志。') + + print('脚本执行完毕,退出。') + exit() From 6df89d31b65aafb6a0bab97c8112c2368205430e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E8=87=A3=E8=B6=85?= <517024110@qq.com> Date: Mon, 28 Sep 2026 18:56:13 +0800 Subject: [PATCH 03/13] =?UTF-8?q?=E9=87=8D=E6=9E=84:=20=E9=A3=8E=E6=89=87?= =?UTF-8?q?=E8=BD=AC=E9=80=9F=E4=B8=8E=E6=B8=A9=E5=BA=A6=E8=AF=BB=E5=8F=96?= =?UTF-8?q?=E7=BB=9F=E4=B8=80=E8=B5=B0=20Prometheus=EF=BC=8C=E6=96=B0?= =?UTF-8?q?=E5=A2=9E=E6=B0=B8=E6=93=8E=20EPYCD8=20=E6=8E=A7=E5=88=B6?= =?UTF-8?q?=E5=99=A8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 基类统一「读」: get_fan_rotational_speed / get_cpu_temperature 收归基类, 经 utils/prometheus_client.py 查询(ipmi_exporter / node_exporter / DCGM) - 各机型子类只保留「写」: set_fan_speed / 手动与自动模式切换 - 新增 fanController/epycd8_controller.py: 永擎 EPYCD8 的 8 字节全量写 (raw 0x3a 0x01,0x00 即交回 BMC 自动),温度源覆盖为 GPU 温度(DCGM) - 入口增加 CONTROLLER_TYPES 机型绑定表;配置模板与 CLAUDE.md 同步 - 修复: max(空列表) ValueError 空值保护 --- CLAUDE.md | 70 ++++++++-- fanController/base_controller.py | 177 +++++++++++++++++++++++-- fanController/dell730_controller.py | 45 ++----- fanController/epycd8_controller.py | 195 ++++++++++++++++++++++++++++ fan_settings.yaml.template | 59 +++++++++ fancontroller.py | 41 ++++-- fancontroller_once.py | 37 ++++-- utils/prometheus_client.py | 190 +++++++++++++++++++++++++++ 8 files changed, 735 insertions(+), 79 deletions(-) create mode 100644 fanController/epycd8_controller.py create mode 100644 utils/prometheus_client.py diff --git a/CLAUDE.md b/CLAUDE.md index 00a9dbc..7e3fb64 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -120,17 +120,25 @@ cat logs/fancontroller.log.YYYY-MM-DD ## 添加新服务器型号支持 -1. 在 `fanController/` 目录下创建新的控制器类文件 (如 `dellr410_controller.py`) -2. 继承 `IPMIFanController` 基类并实现以下方法: - - `get_cpu_temperature()`: 解析 IPMI 温度传感器输出 +> **架构原则(2026-09-28 改造后):「读」统一,「写」分机型。** +> 风扇转速与温度都由基类从 Prometheus 取(数据源是 ipmi_exporter / DCGM), +> 子类只负责实现该机型**特有的「写」命令**。 + +1. 在 `fanController/` 目录下创建新的控制器类文件(如 `epycd8_controller.py`) +2. 继承 `IPMIFanController` 基类,实现该机型特有的写操作: - `set_fan_speed(fan_index, percentage)`: 设置风扇转速的 IPMI raw 命令 - - `get_fan_rotational_speed()`: 解析 IPMI 风扇转速输出 - - `process_server_loop()`: 重写循环模式(如需特定初始化) - - `process_server_once()`: 重写单次模式(如需特定初始化) - - `start_fan_control()`: 启动循环控制 - - `run_once()`: 执行单次控制 -3. (可选) 重写 `set_ipmi_manual_mode()` 和 `set_ipmi_auto_mode()` 如果命令不同 -4. 在 `fancontroller.py` 和 `fancontroller_once.py` 中添加新类型的判断逻辑 + - `_get_base_command()`: 生成基础命令前缀(`ip: "local"` 时返回空串) + - `set_ipmi_manual_mode()` / `set_ipmi_auto_mode()`: 手动/自动模式切换 + (部分机型如 EPYCD8 没有独立的「切手动」命令,写占空比本身就是手动) + - `_build_temperature_query()` (**可选**): 换温度源时覆盖它。 + 默认查 **CPU 核心温度**(node_exporter 的 hwmon,按语义标签 `Tctl` 过滤); + EPYCD8 覆盖此方法去查 **GPU 温度**(DCGM),因为它的机箱风扇是给 GPU 散热的 + - `start_fan_control()` / `run_once()`: 入口方法 +3. **不需要实现 `get_fan_rotational_speed()` 或 `get_cpu_temperature()`** + —— 两者都已由基类统一从 Prometheus 获取(原先各机型各写一份 sdr 解析, + 列结构还互不相同) +4. 在 `fancontroller.py` 与 `fancontroller_once.py` 的 `CONTROLLER_TYPES` + 字典里登记一行:`'新机型名': 新控制器类` ## 重要注意事项 @@ -152,9 +160,48 @@ cat logs/fancontroller.log.YYYY-MM-DD - 自动模式 IPMI 命令: `raw 0x30 0x30 0x01 0x01` - 设置风扇转速: `raw 0x30 0x30 0x02 0x{fan_index:02x} 0x{percentage:02x}` +### Prometheus 数据源(2026-09-28 改造) + +**「读」全部统一走 Prometheus**,不再由各机型子类执行 `ipmitool sdr`: + +```yaml +prometheus: + base_url: "http://192.168.6.31:30091" + timeout: 10 + +servers: + - type: epycd8 + prometheus: # per-server 覆盖(多机场景必需) + fan_instance: "192.168.6.7:9290" # 风扇转速 ← ipmi_exporter + temp_instance: "192.168.6.7:9100" # CPU 温度 ← node_exporter + gpu_instance: "192.168.6.7:9400" # GPU 温度 ← DCGM exporter +``` + +| 数据 | 指标 | instance(pve02 实测) | +|------|------|----------------------| +| 风扇转速 | `ipmi_fan_speed_rpm{name="FRNT_FAN1"}` | `192.168.6.7:9290` | +| CPU 温度 | `node_hwmon_temp_celsius`(按语义标签 `label="Tctl"` 过滤) | `192.168.6.7:9100` | +| GPU 温度 | `DCGM_FI_DEV_GPU_TEMP` | `192.168.6.7:9400` | + +⚠️ **三个 instance 对应三个不同的 exporter,混用会直接查不到数据。** 配置项 +分别为 `fan_instance` / `temp_instance` / `gpu_instance`(笼统的 `instance` 仍 +作兜底)。 + +⚠️ **拿到的是上一次 scrape 的快照,不是实时值。** pve 各 target 的 +`scrape_interval` 实测为 **30s**,也就是温度最多滞后 30 秒 —— 对控速决策是明显 +滞后,**建议把这些 target 的抓取间隔调小**(node_exporter 采集很轻,10s 毫无压力)。 + +⚠️ Prometheus 不可用时读不到任何数据(本地 in-band 的 ipmitool 反而不依赖它)。 +调用方已做空值保护,但排查时先查 Prometheus 连通性。 + +> 补充:CPU 温度走 `node_exporter --collector.hwmon`,它读的就是内核 hwmon +> (`/sys/class/hwmon/`),与直接读 sysfs 是同一份数据。注意 hwmon 里 k10temp +> 的 chip 名是 PCI 路径形式(`pci0000:00_0000:00:18_3`)而非可读的 `k10temp`, +> 所以查询用 `node_hwmon_sensor_label{label="Tctl"}` 做语义过滤,别硬编码 chip 名。 + ### 依赖项 - **PyYAML**: 用于解析 YAML 配置文件 -- 其他功能仅依赖 Python 标准库 +- 其他功能仅依赖 Python 标准库(Prometheus 查询用 `urllib` 实现,无需 `requests`) ## 兼容的服务器型号 @@ -162,3 +209,4 @@ cat logs/fancontroller.log.YYYY-MM-DD |------|------|------------| | Dell | 730XD | `dell730` | | Dell | 730 | `dell730` | +| ASRock Rack | EPYCD8 | `epycd8` | diff --git a/fanController/base_controller.py b/fanController/base_controller.py index c62edae..5ad4d28 100644 --- a/fanController/base_controller.py +++ b/fanController/base_controller.py @@ -4,8 +4,12 @@ import logging from datetime import datetime +from utils.prometheus_client import PrometheusClient, PrometheusError + + class IPMIFanController: - def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True, alert_config=None): + def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True, + alert_config=None, prometheus_config=None): """ 初始化 IPMI 风扇控制器。 @@ -16,6 +20,9 @@ def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True, logger (logging.Logger): 配置好的日志记录器实例。 auto (bool): 是否自动模式,True为自动模式,False为手动模式。 alert_config (dict, optional): 告警配置字典。 + prometheus_config (dict, optional): Prometheus 查询配置,形如 + ``{'base_url': 'http://192.168.6.31:30091', 'instance': '192.168.6.7:9290'}``。 + 风扇转速统一从这里取(见 :meth:`get_fan_rotational_speed`)。 """ self.platform_system = platform.system() if self.platform_system == 'Windows': @@ -56,6 +63,27 @@ def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True, self.logger.error(f"初始化邮件通知器失败: {str(e)}") self.alert_enabled = False + # --- Prometheus 数据源(风扇转速统一从这里取,见 get_fan_rotational_speed)--- + # 全局配置 + servers 项内的 per-server 覆盖:多机场景下每台机器的 + # instance 标签不同,必须逐台指定,否则会把别的机器的转速当成自己的。 + self.prometheus_config = dict(prometheus_config or {}) + server_override = self.servers.get('prometheus') or {} + if server_override: + self.prometheus_config.update(server_override) + + self.prometheus_client = None + base_url = self.prometheus_config.get('base_url') + if base_url: + self.prometheus_client = PrometheusClient( + base_url=base_url, + timeout=self.prometheus_config.get('timeout', 10), + ) + self.logger.info(f"服务器 {self.ip}: 风扇转速数据源 = Prometheus ({base_url})") + else: + self.logger.warning( + f"服务器 {self.ip}: 未配置 prometheus.base_url,风扇转速将无法获取" + ) + def send_command(self, cmd_in): """ 发送命令到系统 Shell。 @@ -96,14 +124,94 @@ def set_fan_speed(self, fan_index, percentage): """ raise NotImplementedError("Method set_fan_speed must be implemented by subclasses") - def get_cpu_temperature(self): + def _pick_instance(self, *keys): + """从 prometheus 配置里按顺序取第一个非空的 instance 标签。 + + ⚠️ **不同数据源的 instance 是不同的**,因为它们是各自的 exporter: + + ============ ================== ========================== + 数据 指标 instance(pve02 实测) + ============ ================== ========================== + 风扇转速 ipmi_fan_speed_rpm ``192.168.6.7:9290`` + CPU 温度 node_hwmon_temp_* ``192.168.6.7:9100`` + GPU 温度 DCGM_FI_DEV_* ``192.168.6.7:9400`` + ============ ================== ========================== + + 混用一个 ``instance`` 会直接查不到数据 —— 这个坑 2026-09-28 实现时踩到。 + 配置里推荐分别写 ``fan_instance`` / ``temp_instance`` / ``gpu_instance``; + 为兼容单数据源场景,仍接受笼统的 ``instance`` 作为兜底。 """ - 获取服务器的 CPU 温度。 + for key in keys: + value = self.prometheus_config.get(key) + if value: + return value + return None - Raises: - NotImplementedError: 子类必须实现此方法。 + #: 温度查询里用作语义过滤的传感器标签。 + #: AMD ``k10temp`` 与 Intel ``coretemp`` 的 CPU 核心温度标签都是 ``Tctl``。 + #: 子类可覆盖本常量,或直接覆盖 :meth:`_build_temperature_query`。 + TEMPERATURE_SENSOR_LABEL = "Tctl" + + def _build_temperature_query(self): + """组装温度查询语句(PromQL)。子类可覆盖以换温度源。 + + 默认查 **CPU 核心温度**,且刻意用**语义标签** ``Tctl`` 过滤,而不是 + 按 hwmon 的 chip 名 —— k10temp 在 Prometheus 里的 chip 名是 PCI 路径 + 形式(``pci0000:00_0000:00:18_3``),硬编码它换台机器就失效了。 + 用 chip 名反查的那个坑 2026-09-28 实测踩过一次。 + + Returns: + str | None: PromQL 语句;返回 ``None`` 表示无法构造(缺配置)。 """ - raise NotImplementedError("Method get_cpu_temperature must be implemented by subclasses") + instance = self._pick_instance('temp_instance', 'instance') + selectors = [f'label="{self.TEMPERATURE_SENSOR_LABEL}"'] + if instance: + selectors.append(f'instance="{instance}"') + label_filter = "{" + ",".join(selectors) + "}" + return ( + "node_hwmon_temp_celsius * on(chip, sensor) group_left(label) " + f"node_hwmon_sensor_label{label_filter}" + ) + + def get_cpu_temperature(self): + """获取温度(°C 列表)—— 统一走 Prometheus,与机型无关。 + + 2026-09-28 改造:与风扇转速同样的思路,「读」这一层收归基类。 + 数据源是 **node_exporter 的 hwmon collector**(``node_hwmon_temp_celsius``) + —— 它读的就是内核 hwmon(``/sys/class/hwmon/``),与直接读 sysfs + 是同一份数据,这里只是换成了 Prometheus 指标这一层封装。 + + ⚠️ **延迟提醒**:拿到的是 Prometheus 上一次 scrape 的快照。pve 各 + target 的 ``scrape_interval`` 实测为 **30s**,也就是温度最多滞后 30 秒。 + 对控速决策而言这是明显滞后 —— 建议把这些 target 的抓取间隔调小, + node_exporter 采集很轻,调到 10s 毫无压力。 + + Returns: + list: 温度列表(°C)。取不到时返回**空列表**,不抛异常。 + """ + if self.prometheus_client is None: + self.logger.error( + f"服务器 {self.ip}: 未配置 Prometheus 数据源,无法获取温度" + ) + return [] + + promql = self._build_temperature_query() + if not promql: + return [] + + try: + samples = self.prometheus_client.query(promql) + except PrometheusError as e: + self.logger.error(f"服务器 {self.ip}: 从 Prometheus 获取温度失败: {e}") + return [] + + temperatures = [s.value for s in samples] + if not temperatures: + self.logger.warning( + f"服务器 {self.ip}: Prometheus 中没有温度数据(查询: {promql})—— " + f"确认目标机的 node_exporter 已开启 --collector.hwmon 并接入 Prometheus" + ) + return temperatures def set_ipmi_manual_mode(self): """ @@ -116,12 +224,48 @@ def set_ipmi_manual_mode(self): return self.ipmi_command(command) def get_fan_rotational_speed(self): - """获取 Dell 730 服务器 风扇转速 的方法。 + """获取风扇转速 —— 统一走 Prometheus(数据源是 ipmi_exporter)。 + + 2026-09-28 改造说明:原先「读转速」下放到各机型子类,各自执行 + ``ipmitool sdr type fan`` 再解析文本。三个问题: + + 1. **输出格式各机型不一致** —— Dell 与 ASRock Rack 的 sdr 列结构完全 + 两套,每加一个机型就要重写一遍解析(还容易写错,见 sensors.py 里 + 那个把传感器 ID 当成 RPM 的踩坑记录) + 2. **每轮都要 spawn 一个 ipmitool 进程**,还要管 BMC 连接 + 3. **读数口径可能与看板对不上** —— 控制器一个值、Grafana 另一个值 + + 改成统一查 Prometheus 后,「读」这一层与机型无关了,子类只需负责 + 「怎么写」(见 :meth:`set_fan_speed`)。 + + ⚠️ **代价**:拿到的是上一次 scrape 的快照(延迟由 Prometheus 的 + ``scrape_interval`` 决定),且 Prometheus 不可用时完全读不到数据。 Returns: - list: 包含 风扇转速 的列表。 + list: 风扇转速(RPM)列表。取不到时返回**空列表**,不抛异常 —— + 调用方必须自行处理空结果。 """ - raise NotImplementedError("Method get_cpu_temperature must be implemented by subclasses") + if self.prometheus_client is None: + self.logger.error( + f"服务器 {self.ip}: 未配置 Prometheus 数据源,无法获取风扇转速" + ) + return [] + + try: + speeds = self.prometheus_client.get_fan_speeds( + instance=self._pick_instance('fan_instance', 'instance') + ) + except PrometheusError as e: + self.logger.error(f"服务器 {self.ip}: 从 Prometheus 获取风扇转速失败: {e}") + return [] + + if not speeds: + self.logger.warning( + f"服务器 {self.ip}: Prometheus 中没有风扇转速数据 —— " + f"请确认该机器的 ipmi_exporter 已接入,且 prometheus.instance " + f"标签配置正确" + ) + return speeds def adjust_fans_once(self, prev_temp_ranges=None, prev_fan_speeds=None): """ @@ -154,8 +298,19 @@ def adjust_fans_once(self, prev_temp_ranges=None, prev_fan_speeds=None): max_temp_value = max(cpu_temps) min_temp_value = min(cpu_temps) avg_temp_value = sum(cpu_temps) // len(cpu_temps) - max_fan_speed = max(current_fan_speeds) - min_fan_speed = min(current_fan_speeds) + # ⚠️ 空值保护:改用 Prometheus 取数后,「查不到数据」是正常情况 + # (exporter 未接入 / Prometheus 挂了 / instance 标签配错)。 + # 上游在这里直接 max([]) 会 ValueError 把整个控制线程打挂, + # 进程还活着但已经不再控风扇 —— 典型的静默失效。 + if current_fan_speeds: + max_fan_speed = max(current_fan_speeds) + min_fan_speed = min(current_fan_speeds) + else: + max_fan_speed = 0 + min_fan_speed = 0 + self.logger.warning( + f"服务器 {self.ip}: 本轮没有风扇转速数据,跳过转速相关判断" + ) result['cpu_temp'] = max_temp_value result['max_fan_speed'] = max_fan_speed diff --git a/fanController/dell730_controller.py b/fanController/dell730_controller.py index 2a93a77..1d43e27 100644 --- a/fanController/dell730_controller.py +++ b/fanController/dell730_controller.py @@ -1,4 +1,3 @@ -import re import time import logging @@ -28,41 +27,15 @@ def set_fan_speed(self, fan_index, percentage): set_speed_cmd = f"{base_cmd} raw 0x30 0x30 0x02 0x{fan_index:02x} 0x{hex_percentage}" self.ipmi_command(set_speed_cmd.strip()) - def get_cpu_temperature(self): - """获取 Dell 730 服务器 CPU 温度的方法。 - - Returns: - list: 包含 CPU 温度的列表。 - """ - base_cmd = self._get_base_command() - command = f"{base_cmd} sdr type Temperature" - output = self.ipmi_command(command.strip()) - - temp_list = [] - for line in output.split("\n"): - items = line.split("|") - if len(items) > 1 and "ok" in items[2]: - if "0Eh" in items[1] or "0Fh" in items[1]: - temp_match = re.search(r'(\d+)\s+degrees\s+C', items[4]) - if temp_match: - temp = int(temp_match.group(1)) - temp_list.append(temp) - return temp_list - - def get_fan_rotational_speed(self): - """获取 Dell 730 服务器 风扇转速 的方法。 - - Returns: - list: 包含 风扇转速 的列表。 - """ - base_cmd = self._get_base_command() - command = f"{base_cmd} sdr type fan" - output = self.ipmi_command(command.strip()) - - rpm_values = re.findall(r'\|\s(\d+)\sRPM', output) - rpm_values = [int(rpm) for rpm in rpm_values] - - return rpm_values + # 注意:get_cpu_temperature() 与 get_fan_rotational_speed() 已于 2026-09-28 + # 从本类**移除**,「读」这一层收归基类统一走 Prometheus: + # - 温度 ← node_hwmon_temp_celsius(node_exporter 的 hwmon collector) + # - 转速 ← ipmi_fan_speed_rpm(ipmi_exporter) + # 原先这两个方法各自执行 `ipmitool sdr type Temperature / fan` 再正则解析, + # 而各机型的 sdr 列结构完全不同(Dell 与 ASRock Rack 就是两套),属于 + # 「按机型重复实现同一件事」。旧的 CPU 温度解析还依赖 `0Eh` / `0Fh` 这种 + # 传感器 ID 硬编码,固件一升级就可能失效。 + # 本类现在只负责 Dell 特有的「写」:set_fan_speed / 手动模式 / PCIe 散热响应。 def _initialize_dell730(self): """ diff --git a/fanController/epycd8_controller.py b/fanController/epycd8_controller.py new file mode 100644 index 0000000..ad88be2 --- /dev/null +++ b/fanController/epycd8_controller.py @@ -0,0 +1,195 @@ +"""ASRock Rack EPYCD8(永擎)风扇控制器。 + +**与 Dell 系列的区别全在「写」这一侧** —— 读取(温度、转速)由基类统一 +从 Prometheus 取,与本类无关: + +================= ======================================== ============================== +机型 设置风扇 恢复自动 +================= ======================================== ============================== +Dell 730 ``raw 0x30 0x30 0x02 0x{idx} 0x{pct}`` ``raw 0x30 0x30 0x01 0x01`` + 逐个风扇单独设,且**必须先切手动模式** +EPYCD8 ``raw 0x3a 0x01 b1..b8`` ``raw 0x3a 0x01 0x00 × 8`` + **一次全量写 8 字节**,无独立手动模式命令 +================= ======================================== ============================== + +**EPYCD8 的坑**(2026-09-17 / 09-28 实测): + +1. **必须写满 8 个字节。** 少写一个字节 BMC 不报错、返回码仍为 0,但转速 + 纹丝不动。实测曾误给 7 字节,一度误判「该风扇位不可控」。 +2. **``0x00`` 就是「交回 BMC 自动」** —— 所以本机型不需要单独的 auto 命令, + 退出时把 8 字节全填 0x00 即可,比 Dell 省事。 +3. **手动值不持久化**,BMC 重启或整机断电后失效,自动回到 BMC 自动策略。 +4. CPU 温度达到临界阈值时,BMC 会**强行覆盖**手动值(热保护,不可对抗, + 也不应尝试对抗)。 + +8 字节位映射(b2 为保留位,恒 ``0x00``):: + + b1 CPU1_FAN1 + b2 --(保留) + b3 REAR_FAN1 + b4 REAR_FAN2 ← pve02 用于 Tesla T10 散热 + b5 FRNT_FAN1 ← pve02 用于 Tesla T10 散热 + b6 FRNT_FAN2 (pve02 未接风扇) + b7 FRNT_FAN3 (pve02 未接风扇) + b8 FRNT_FAN4 (pve02 未接风扇) +""" + +from .base_controller import IPMIFanController + + +class Epycd8FanController(IPMIFanController): + """ASRock Rack EPYCD8 风扇控制器。 + + 只负责 EPYCD8 特有的「写」;「读」由基类提供(风扇转速走 Prometheus, + 温度在本类重写为取 GPU 温度,原因见 :meth:`get_cpu_temperature`)。 + """ + + #: 8 字节 payload 长度 —— 硬性要求,少一个字节整条命令会被静默忽略 + PAYLOAD_LEN = 8 + + #: 风扇位名称 → payload 下标(0-based)。基类的 ``enumerate(fan_speeds)`` + #: 传进来的 ``fan_index`` 就是这个下标。 + FAN_SLOT_INDEX = { + 'CPU1_FAN1': 0, + 'REAR_FAN1': 2, + 'REAR_FAN2': 3, + 'FRNT_FAN1': 4, + 'FRNT_FAN2': 5, + 'FRNT_FAN3': 6, + 'FRNT_FAN4': 7, + } + + #: 下标 → 风扇位名(构造一次,省得每次反查) + _INDEX_TO_SLOT = {v: k for k, v in FAN_SLOT_INDEX.items()} + + #: payload 中保留位的下标 + RESERVED_INDEX = 1 + + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + # 维护 8 字节目标状态。BMC 要求全量写,所以必须记住其它位设过什么, + # 否则「只想改一个位」会把别的位一起踩成 0x00(= 全部交回自动)。 + # 初值全 0x00 = 全部交回 BMC 自动,是安全的起点。 + self._fan_duties = {slot: 0x00 for slot in self.FAN_SLOT_INDEX} + + # ------------------------------------------------------------ 基础命令 + + def _get_base_command(self): + """根据 IP 配置生成基础 IPMI 命令。 + + ``ip: "local"`` 时走本地 in-band(``/dev/ipmi0``),不带 lanplus 参数。 + """ + if self.ip == 'local': + return "" + return f"-I lanplus -H {self.ip} -U {self.user} -P {self.password}" + + # ------------------------------------------------------------ 读取(重写) + + def _build_temperature_query(self): + """覆盖基类:EPYCD8 要的是 **GPU 温度**,不是 CPU 温度。 + + 原因:BMC 里没有任何 GPU 温度传感器(实测温度项只有 MB / Card Side / + CPU / TR1 / DDR4_A~H),而机箱风扇 FRNT_FAN1 / REAR_FAN2 是两块 + Tesla T10(原厂被动散热)的唯一散热手段。所以基类里「CPU 温度」这个 + 字段名,在本机型上承载的其实是 GPU 温度。 + + 数据源同样是 Prometheus —— DCGM exporter 已接入 + (job ``dcgm-exporter-pve02``),指标 ``DCGM_FI_DEV_GPU_TEMP``。 + + Returns: + str | None: PromQL;未配置 ``gpu_instance`` 时返回 ``None``。 + """ + instance = self.prometheus_config.get('gpu_instance') + if not instance: + self.logger.error( + f"服务器 {self.ip}: 未配置 prometheus.gpu_instance,无法定位 " + f"DCGM 数据(例: 192.168.6.7:9400)" + ) + return None + return f'DCGM_FI_DEV_GPU_TEMP{{instance="{instance}"}}' + + # ------------------------------------------------------------ 写入 + + def set_fan_speed(self, fan_index, percentage): + """设置风扇转速。 + + Args: + fan_index (int): **payload 下标**(0-based),与配置里 ``fan_speeds`` + 列表的位置一一对应。 + percentage (int): 转速百分比。``1~100`` = 手动占空比; + **``0`` = 把该位交回 BMC 自动控制**(这和 Dell 不同 —— + Dell 那边 0 无意义,EPYCD8 这边 0 是个有语义的值)。 + + 说明:单次调用**不会**立即下发,而是更新目标状态后整包写出。 + 因为 BMC 要求 8 字节全量写,逐个位调用来回写 8 次既慢又容易互相踩。 + 基类的 ``adjust_fans_once`` 会 ``enumerate`` 遍历所有位,所以 + 一整套调完正好写一整包。 + """ + slot = self._INDEX_TO_SLOT.get(fan_index) + if slot is None: + self.logger.warning( + f"服务器 {self.ip}: 未知风扇下标 {fan_index}(合法范围 " + f"0~{self.PAYLOAD_LEN - 1}),跳过" + ) + return + + if percentage == 0: + duty = 0x00 + elif 1 <= percentage <= 100: + duty = percentage + else: + self.logger.warning( + f"服务器 {self.ip}: 非法占空比 {percentage}%(应为 0 或 1~100),跳过" + ) + return + + self._fan_duties[slot] = duty + self._write_payload() + + def _write_payload(self): + """把当前目标状态按 8 字节全量写下去。 + + **永远写满 8 字节** —— 这是这个机型最容易踩的坑,见模块头部说明。 + """ + payload = [0x00] * self.PAYLOAD_LEN + for slot, duty in self._fan_duties.items(): + payload[self.FAN_SLOT_INDEX[slot]] = duty + payload[self.RESERVED_INDEX] = 0x00 # 保留位恒 0 + + bytes_str = " ".join(f"0x{b:02x}" for b in payload) + base_cmd = self._get_base_command() + command = f"{base_cmd} raw 0x3a 0x01 {bytes_str}" + self.ipmi_command(command.strip()) + + # ------------------------------------------------------------ 模式切换 + + def set_ipmi_manual_mode(self): + """EPYCD8 没有独立的「切手动」命令 —— 写入非零占空比本身就是手动模式。 + + 所以这里**什么都不做**,只记一条日志。刻意不去刷一遍 payload: + 程序刚启动、目标值还没算出来的时候刷一遍,会把当前状态清成全自动, + 造成一次没必要的转速波动。 + """ + self.logger.info( + f"服务器 {self.ip}: EPYCD8 无需单独切换手动模式(写占空比即进入手动)" + ) + + def set_ipmi_auto_mode(self): + """把全部风扇位交回 BMC 自动控制(8 字节全填 0x00)。 + + 这是本机型的**安全回退动作**,也是它比 Dell 省事的地方 —— + 不需要额外的 ``raw 0x30 0x30 0x01 0x01``。 + """ + self._fan_duties = {slot: 0x00 for slot in self.FAN_SLOT_INDEX} + self._write_payload() + self.logger.info(f"服务器 {self.ip}: 已把全部风扇位交回 BMC 自动控制") + + # ------------------------------------------------------------ 入口 + + def start_fan_control(self): + """启动循环控制(供 ``fancontroller.py`` 调用)。""" + self.process_server_loop() + + def run_once(self): + """执行一次控制(供 ``fancontroller_once.py`` 调用)。""" + self.process_server_once() diff --git a/fan_settings.yaml.template b/fan_settings.yaml.template index 8b4d92e..8881c77 100644 --- a/fan_settings.yaml.template +++ b/fan_settings.yaml.template @@ -45,6 +45,23 @@ alert: - "admin@example.com" - "alert@example.com" +# Prometheus 数据源(风扇转速统一从这里取) +# +# 2026-09-28 改造:不再由各机型子类执行 `ipmitool sdr type fan` 拉转速, +# 统一改为查 Prometheus(数据源是 ipmi_exporter)。这样「读」这一层与机型 +# 无关 —— 加新机型时只需要写它特有的「写」命令。 +# +# ⚠️ 注意:拿到的是**上一次 scrape 的快照**,不是实时值。所以「设完转速立刻 +# 回读验证」会读到旧值,需要等一个 scrape 周期。 +prometheus: + # Prometheus 地址(本项目实测为 k8s 集群 NodePort) + base_url: "http://192.168.6.31:30091" + # 单次查询超时(秒) + timeout: 10 + # + # 「读哪台机器」由 instance 标签决定,**必须逐台在 servers[].prometheus + # 里指定** —— 不同数据的 instance 是不同的 exporter 地址,见下面 epycd8 示例。 + # 服务器列表 servers: - type: dell730 # 服务器类型 @@ -82,3 +99,45 @@ servers: - min_temp: 61 max_temp: 80 fan_speeds: [25, 25, 25, 25, 25, 25] + + # ================= ASRock Rack EPYCD8(永擎)================= + # + # 与 Dell 的关键差异(写命令完全不同,详见 fanController/epycd8_controller.py): + # + # 1. fan_speeds 必须给 **8 个值**(Dell 730 是 6 个),按 payload 下标对应: + # 下标 0 = CPU1_FAN1 下标 1 = 保留位(恒自动) + # 下标 2 = REAR_FAN1 下标 3 = REAR_FAN2 + # 下标 4 = FRNT_FAN1 下标 5~7 = FRNT_FAN2~4 + # + # 2. **值写 0 = 把该位交回 BMC 自动控制**(Dell 那边 0 没有意义)。 + # 所以「只控两个 GPU 风扇位」就写成 [0,0,0,50,50,0,0,0] + # + # 3. 本机型的「温度」指的是 **GPU 温度**:BMC 里没有任何 GPU 温度传感器, + # 而机箱风扇是 Tesla T10(原厂被动散热)的唯一散热手段。温度源同样 + # 走 Prometheus(DCGM exporter 的 DCGM_FI_DEV_GPU_TEMP)。 + - type: epycd8 + ip: "local" # 脚本跑在目标机本机时填 local(走 in-band,无需凭据) + user: "-" + password: "-" + prometheus: # per-server 覆盖全局配置(多机场景必需) + # ⚠️ 三个 instance 对应三个**不同的 exporter**,混用会直接查不到数据: + fan_instance: "192.168.6.7:9290" # 风扇转速 ← ipmi_exporter + temp_instance: "192.168.6.7:9100" # CPU 温度 ← node_exporter(需开 --collector.hwmon) + gpu_instance: "192.168.6.7:9400" # GPU 温度 ← DCGM exporter(仅本机型用) + temperature_ranges: + # 温度区间按 GPU 温度划分(多卡时取最热那张) + - min_temp: 0 + max_temp: 54 + fan_speeds: [0, 0, 0, 40, 40, 0, 0, 0] # REAR_FAN2 + FRNT_FAN1 = 40% + - min_temp: 55 + max_temp: 64 + fan_speeds: [0, 0, 0, 50, 50, 0, 0, 0] + - min_temp: 65 + max_temp: 74 + fan_speeds: [0, 0, 0, 70, 70, 0, 0, 0] + - min_temp: 75 + max_temp: 84 + fan_speeds: [0, 0, 0, 85, 85, 0, 0, 0] + - min_temp: 85 + max_temp: 120 + fan_speeds: [0, 0, 0, 100, 100, 0, 0, 0] # 拉满 diff --git a/fancontroller.py b/fancontroller.py index e6bb6d9..17edcad 100644 --- a/fancontroller.py +++ b/fancontroller.py @@ -22,6 +22,16 @@ import yaml from fanController.dell730_controller import Dell730FanController +from fanController.epycd8_controller import Epycd8FanController + + +#: 机型 → 控制器类 的映射。 +#: 配置里 ``servers[].type`` 填什么,就分发到哪个控制器 —— 新增机型时在这里 +#: 登记一行即可,下面的分发逻辑不用动。 +CONTROLLER_TYPES = { + 'dell730': Dell730FanController, + 'epycd8': Epycd8FanController, +} def main(): @@ -79,21 +89,30 @@ def main(): windows_ipmi_tool_path = data['windows_ipmi_tool_path'] interval = data['interval'] alert_config = data.get('alert', {}) # 获取告警配置 + prometheus_config = data.get('prometheus', {}) # Prometheus 数据源配置 threads = [] for server in servers: - if server['type'] == 'dell730': - fan_controller = Dell730FanController( - servers=server, - interval=interval, - windows_ipmi_tool_path=windows_ipmi_tool_path, - logger=logger, - auto=True, # 循环模式 - alert_config=alert_config + controller_class = CONTROLLER_TYPES.get(server['type']) + if controller_class is None: + logger.warning( + f"未知的服务器类型 {server['type']!r}({server.get('ip')}),已跳过。" + f"当前支持的机型: {', '.join(sorted(CONTROLLER_TYPES))}" ) - thread = threading.Thread(target=fan_controller.start_fan_control, name=f"Thread-{server['ip']}") - thread.start() - threads.append(thread) + continue + + fan_controller = controller_class( + servers=server, + interval=interval, + windows_ipmi_tool_path=windows_ipmi_tool_path, + logger=logger, + auto=True, # 循环模式 + alert_config=alert_config, + prometheus_config=prometheus_config, + ) + thread = threading.Thread(target=fan_controller.start_fan_control, name=f"Thread-{server['ip']}") + thread.start() + threads.append(thread) for thread in threads: thread.join() diff --git a/fancontroller_once.py b/fancontroller_once.py index c112ede..e6e66d0 100644 --- a/fancontroller_once.py +++ b/fancontroller_once.py @@ -21,6 +21,14 @@ import yaml from fanController.dell730_controller import Dell730FanController +from fanController.epycd8_controller import Epycd8FanController + + +#: 机型 → 控制器类 的映射(与 fancontroller.py 保持一致)。 +CONTROLLER_TYPES = { + 'dell730': Dell730FanController, + 'epycd8': Epycd8FanController, +} def main(): @@ -76,19 +84,28 @@ def main(): windows_ipmi_tool_path = data['windows_ipmi_tool_path'] interval = data.get('interval', 60) # 单次模式不使用 interval,但保留参数兼容性 alert_config = data.get('alert', {}) # 获取告警配置 + prometheus_config = data.get('prometheus', {}) # Prometheus 数据源配置 for server in servers: - if server['type'] == 'dell730': - fan_controller = Dell730FanController( - servers=server, - interval=interval, - windows_ipmi_tool_path=windows_ipmi_tool_path, - logger=logger, - auto=False, # 单次模式不需要 auto 参数 - alert_config=alert_config + controller_class = CONTROLLER_TYPES.get(server['type']) + if controller_class is None: + logger.warning( + f"未知的服务器类型 {server['type']!r}({server.get('ip')}),已跳过。" + f"当前支持的机型: {', '.join(sorted(CONTROLLER_TYPES))}" ) - # 执行一次风扇控制 - fan_controller.run_once() + continue + + fan_controller = controller_class( + servers=server, + interval=interval, + windows_ipmi_tool_path=windows_ipmi_tool_path, + logger=logger, + auto=False, # 单次模式不需要 auto 参数 + alert_config=alert_config, + prometheus_config=prometheus_config, + ) + # 执行一次风扇控制 + fan_controller.run_once() logger.info("单次执行完成") diff --git a/utils/prometheus_client.py b/utils/prometheus_client.py new file mode 100644 index 0000000..2badf04 --- /dev/null +++ b/utils/prometheus_client.py @@ -0,0 +1,190 @@ +"""Prometheus HTTP API 查询封装。 + +**背景**:原先风扇转速是直接执行 ``ipmitool sdr type fan`` 再解析文本得到的。 +2026-09-28 起改为统一走 Prometheus(数据源是 ipmi_exporter)。 + +这么改的好处: + +1. 读数口径与看板一致,不会出现「控制器一个值、Grafana 另一个值」 +2. 不必每轮 spawn 一个 ipmitool 进程,也不用管 BMC 连接 +3. ipmitool 的人类可读输出格式在不同机型/版本差异极大(Dell 与 ASRock Rack + 就完全是两套列结构),而 Prometheus 指标格式是标准化的 + +**代价(必须知道,不然会踩坑)**: + +1. 拿到的是**上一次 scrape 的快照**,不是实时值。延迟由 Prometheus 的 + ``scrape_interval`` 决定。所以「设完转速立刻回读验证」这类操作会读到旧值, + 需要等一个 scrape 周期。 +2. **Prometheus 不可用时完全读不到数据** —— 本地 in-band 的 ipmitool 反而 + 不依赖它。所以调用方必须处理空结果,不能假设一定有数据。 +3. 所有被监控机器的 ``ipmi_fan_speed_rpm`` 混在同一个指标里,**必须用 + ``instance`` 标签限定目标机器**,否则会把 A 机器的转速当成 B 机器的。 + +只用标准库实现,不引入 ``requests`` —— 这个项目当前的依赖只有 PyYAML, +保持这一点。 +""" + +from __future__ import annotations + +import json +import logging +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass, field + +logger = logging.getLogger(__name__) + + +class PrometheusError(RuntimeError): + """Prometheus 查询失败(网络不通、返回非法 JSON、查询报错等)。""" + + +@dataclass(frozen=True) +class PromSample: + """一条查询结果样本。""" + + labels: dict = field(default_factory=dict) + value: float = 0.0 + + @property + def name(self) -> str: + """``name`` 标签,即传感器/风扇位名称。""" + return self.labels.get("name", "") + + @property + def instance(self) -> str: + return self.labels.get("instance", "") + + +class PrometheusClient: + """极简 Prometheus HTTP API 客户端(只用到瞬时查询)。""" + + def __init__(self, base_url: str, timeout: float = 10.0): + """ + Args: + base_url: Prometheus 地址,如 ``http://192.168.6.31:30091``。 + timeout: 单次查询超时(秒)。**必须有** —— 上游项目就是因为 + 不带超时的阻塞式 subprocess 读取而静默挂死过。 + """ + self.base_url = base_url.rstrip("/") + self.timeout = timeout + + # ------------------------------------------------------------ 基础查询 + + def query(self, promql: str) -> list[PromSample]: + """执行瞬时查询(``/api/v1/query``)。 + + Raises: + PrometheusError: 网络不通、返回非法 JSON,或查询本身报错。 + """ + params = urllib.parse.urlencode({"query": promql}) + url = f"{self.base_url}/api/v1/query?{params}" + request = urllib.request.Request( + url, headers={"User-Agent": "python-ipmitool/1.0"} + ) + + try: + with urllib.request.urlopen(request, timeout=self.timeout) as response: + payload = json.loads(response.read().decode("utf-8", errors="replace")) + except urllib.error.URLError as exc: + raise PrometheusError(f"连不上 Prometheus {self.base_url}: {exc}") from exc + except (json.JSONDecodeError, ValueError) as exc: + raise PrometheusError(f"Prometheus 返回的不是合法 JSON: {exc}") from exc + + if payload.get("status") != "success": + raise PrometheusError( + f"Prometheus 查询失败: {payload.get('error', payload)}" + ) + + samples: list[PromSample] = [] + for item in payload.get("data", {}).get("result", []): + try: + value = float(item["value"][1]) + except (KeyError, IndexError, ValueError, TypeError): + continue # 非向量结果或 NaN 之类,跳过 + samples.append(PromSample(labels=item.get("metric", {}), value=value)) + return samples + + # ------------------------------------------------------------ 风扇转速 + + def _fan_selector(self, instance: str | None, job: str | None) -> str: + selectors = [] + if instance: + selectors.append(f'instance="{instance}"') + if job: + selectors.append(f'job="{job}"') + return "{" + ",".join(selectors) + "}" if selectors else "" + + def get_fan_speeds_by_name( + self, instance: str | None = None, job: str | None = None + ) -> dict[str, float]: + """查询风扇转速,返回 ``{风扇位名: RPM}``。 + + Args: + instance: 限定 Prometheus 的 ``instance`` 标签,如 + ``192.168.6.7:9290``。**多机环境务必传**,否则会把别的 + 机器的转速混进来。 + job: 可选,进一步限定抓取任务名。 + """ + promql = "ipmi_fan_speed_rpm" + self._fan_selector(instance, job) + samples = self.query(promql) + readings = {s.name: s.value for s in samples if s.name} + if not readings: + logger.warning( + "Prometheus 里查不到风扇转速(查询: %s)—— " + "确认目标机器的 ipmi_exporter 已接入且 instance 标签填对了", + promql, + ) + else: + logger.debug( + "从 Prometheus 取到 %d 个风扇位: %s", + len(readings), + ", ".join(f"{k}={v:.0f}" for k, v in readings.items()), + ) + return readings + + def get_fan_speeds( + self, instance: str | None = None, job: str | None = None + ) -> list[float]: + """查询风扇转速,返回 RPM 列表。 + + 注意:返回的是**列表**,风扇位顺序由 Prometheus 返回顺序决定, + 不保证与机箱物理编号一致。调用方若只关心「最高的那个」(比如 + 上游的异常转速判断)则不受影响;若要按位对应,请改用 + :meth:`get_fan_speeds_by_name`。 + """ + return list(self.get_fan_speeds_by_name(instance, job).values()) + + # ------------------------------------------------------------ GPU 温度 + + def get_gpu_temperatures( + self, instance: str | None = None, job: str | None = None + ) -> list[float]: + """查询 GPU 温度(DCGM),返回 °C 列表。 + + ``DCGM_FI_DEV_GPU_TEMP`` 每张卡一条,靠 ``UUID`` 标签区分。这里只返回 + 温度值列表,调用方通常取 ``max()`` —— 多卡机器上「最热的那张」才是 + 决定风量的那张。 + + Args: + instance: DCGM exporter 的 instance 标签,如 ``192.168.6.7:9400``。 + """ + selectors = [] + if instance: + selectors.append(f'instance="{instance}"') + if job: + selectors.append(f'job="{job}"') + promql = "DCGM_FI_DEV_GPU_TEMP" + if selectors: + promql += "{" + ",".join(selectors) + "}" + + samples = self.query(promql) + temperatures = [s.value for s in samples] + if not temperatures: + logger.warning( + "Prometheus 里查不到 GPU 温度(查询: %s)—— " + "确认 DCGM exporter 已接入且 instance 标签填对了", + promql, + ) + return temperatures From 9319ee0f8d1ee0668318dcaf1ea392530539032a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E8=87=A3=E8=B6=85?= <517024110@qq.com> Date: Mon, 28 Sep 2026 18:56:27 +0800 Subject: [PATCH 04/13] =?UTF-8?q?=E6=96=B0=E5=A2=9E:=20GPU=20=E9=A3=8E?= =?UTF-8?q?=E6=89=87=E6=8E=A7=E5=88=B6=E5=8F=B0=E5=8D=95=E6=9C=BA=E5=BA=94?= =?UTF-8?q?=E7=94=A8=EF=BC=88FastAPI=20+=20=E9=97=AD=E7=8E=AF=E6=8E=A7?= =?UTF-8?q?=E5=88=B6=20+=20=E5=AE=89=E5=85=A8=E6=8A=A4=E6=A0=8F=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit app/ 为跑在 pve02 本机的单机应用(观测走 exporter,调控走 ipmitool raw): - controller.py 闭环控制: 「散热源(GPU/CPU) → 风扇位」分配驱动, 分段曲线 + 降温滞回 + 紧急全速(含滞回解除);分配未配置时兜底 跟随最热卡,不掉进 BMC 失明档 - safety.py 安全护栏: SIGTERM/SIGINT/atexit 全路径交回 BMC 自动, 心跳文件供独立看门狗兜底 - ipmi.py 8 字节全量 payload 编解码(少一字节 BMC 静默忽略) - sensors.py 多源读取: DCGM 端点 / nvidia-smi 兜底 / ipmi_exporter / node_exporter hwmon(k10temp 按语义标签 Tctl 关联) - store.py SQLite 持久化: 分配关系 + 运行时设置 + 操作审计 (谁在什么时候改了风扇,查 audit_log) - api.py REST + WebSocket 实时快照;api/history 代理 Prometheus query_range(浏览器不直连监控栈) - 设置与分配双向打通: 未纳入管控的 GPU 不能持有分配;取消管控 联动清除分配;启动恢复先设置后分配 + 脏数据自愈 - 掉卡语义: 该风扇位交回 BMC 自动(不狂转),卡恢复后自动接管 - deploy/: systemd 服务(开机自启 + TimeoutStopSec 保证退出回落) 与独立心跳看门狗(timer 每 2 分钟检查) - 54 个单元测试(payload 编解码 / 曲线滞回 / 分配校验 / 传感器解析) --- .gitignore | 10 + app/__init__.py | 11 + app/api.py | 483 +++++++++++++ app/config.py | 161 +++++ app/config.yaml | 90 +++ app/controller.py | 1074 ++++++++++++++++++++++++++++ app/curve.py | 164 +++++ app/deploy/fan-watchdog.service | 8 + app/deploy/fan-watchdog.sh | 65 ++ app/deploy/fan-watchdog.timer | 12 + app/deploy/gpu-fan-console.service | 28 + app/diagnose.py | 268 +++++++ app/ipmi.py | 339 +++++++++ app/main.py | 198 +++++ app/requirements.txt | 4 + app/safety.py | 186 +++++ app/sensors.py | 632 ++++++++++++++++ app/store.py | 255 +++++++ app/tests/__init__.py | 1 + app/tests/test_core.py | 624 ++++++++++++++++ 20 files changed, 4613 insertions(+) create mode 100644 app/__init__.py create mode 100644 app/api.py create mode 100644 app/config.py create mode 100644 app/config.yaml create mode 100644 app/controller.py create mode 100644 app/curve.py create mode 100644 app/deploy/fan-watchdog.service create mode 100644 app/deploy/fan-watchdog.sh create mode 100644 app/deploy/fan-watchdog.timer create mode 100644 app/deploy/gpu-fan-console.service create mode 100644 app/diagnose.py create mode 100644 app/ipmi.py create mode 100644 app/main.py create mode 100644 app/requirements.txt create mode 100644 app/safety.py create mode 100644 app/sensors.py create mode 100644 app/store.py create mode 100644 app/tests/__init__.py create mode 100644 app/tests/test_core.py diff --git a/.gitignore b/.gitignore index c86d4cb..1d3f4df 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,13 @@ +# 本地工作数据(WorkBuddy 记忆目录与 agent-ssh-cli 锁文件),不进版本库 +.workbuddy/ +agent-ssh-cli-*.lock + +# 运行时数据:SQLite 库(绑定 + 审计)、心跳文件等,属于机器本地状态 +app/data/ + +# 前端构建产物:由 frontend/ 构建生成(npm run build),不进版本库 +app/static/ + # Byte-compiled / optimized / DLL files __pycache__/ *.py[cod] diff --git a/app/__init__.py b/app/__init__.py new file mode 100644 index 0000000..ac3fb04 --- /dev/null +++ b/app/__init__.py @@ -0,0 +1,11 @@ +"""GPU 风扇监控控制台 —— 单机应用。 + +一个进程同时承担三件事: +1. **控制回路**:读 GPU 温度 → 算占空比 → 写 IPMI(后台任务) +2. **API 服务**:REST + WebSocket 给前端 +3. **静态托管**:直接伺服前端构建产物 + +没有跨机通信、没有独立 agent —— 它只管 pve02 自己这台机器的风扇。 +""" + +__version__ = "0.1.0" diff --git a/app/api.py b/app/api.py new file mode 100644 index 0000000..dab2e12 --- /dev/null +++ b/app/api.py @@ -0,0 +1,483 @@ +"""REST + WebSocket 路由。 + +控制器实例挂在 ``app.state`` 上,路由通过 ``request.app.state`` 取用 —— +不引入全局单例,方便将来写测试时替换成假的控制器。 +""" + +from __future__ import annotations + +import asyncio +import json +import logging +import time +import urllib.error +import urllib.parse +import urllib.request +from typing import Any, Literal + +from fastapi import ( + APIRouter, + HTTPException, + Query, + Request, + WebSocket, + WebSocketDisconnect, +) +from pydantic import BaseModel, Field + +from .controller import FanController + +logger = logging.getLogger(__name__) + +router = APIRouter() + +#: WebSocket 推送间隔(秒)。比控制周期快,前端看起来才像"实时"。 +WS_PUSH_INTERVAL = 2.0 + + +# ------------------------------------------------------------------ 请求模型 + + +class ModeRequest(BaseModel): + mode: Literal["auto", "manual"] = Field(description="auto=曲线自动;manual=手动固定") + + +class ManualDutyRequest(BaseModel): + slot: str = Field(description="风扇位名,如 FRNT_FAN1") + duty: int | None = Field( + default=None, + ge=1, + le=100, + description="占空比百分比。传 null 表示把该位交回 BMC 自动控制。", + ) + + +class AssignmentItem(BaseModel): + """一个散热源的风扇分配。**主体是源(GPU),它去挑风扇接口。**""" + + key: str = Field(description="源的唯一键:gpu: 或 cpu") + kind: Literal["gpu", "cpu"] = Field( + default="gpu", + description="gpu = 某张具体的卡;cpu = CPU 核温度 Tctl", + ) + gpu_uuid: str | None = Field( + default=None, description="kind=gpu 时的 GPU UUID(绝不用 index)" + ) + slots: list[str] = Field( + default_factory=list, description="分配给该源的风扇位(可多选)" + ) + + +class AssignmentsRequest(BaseModel): + assignments: list[AssignmentItem] + + +# ------------------------------------------------------------------ 工具 + + +def _controller(request: Request) -> FanController: + controller = getattr(request.app.state, "controller", None) + if controller is None: # pragma: no cover - 正常启动流程不会走到 + raise HTTPException(status_code=503, detail="控制器尚未初始化") + return controller + + +def _query_range( + base_url: str, promql: str, start: float, end: float, step: float +) -> list[dict[str, Any]]: + """执行一次 Prometheus ``query_range``(阻塞调用,外层丢线程池)。 + + 只返回 ``[{"metric": {...}, "points": [[ms, value], ...]}]``, + 时间戳统一换算成**毫秒** —— ECharts 的 time 轴要毫秒。 + """ + params = urllib.parse.urlencode( + { + "query": promql, + "start": f"{start:.3f}", + "end": f"{end:.3f}", + "step": f"{int(step)}", + } + ) + url = f"{base_url.rstrip('/')}/api/v1/query_range?{params}" + + with urllib.request.urlopen(url, timeout=15) as response: + payload = json.loads(response.read().decode("utf-8", errors="replace")) + + if payload.get("status") != "success": + raise RuntimeError(str(payload.get("error", "query_range 返回失败"))) + + series: list[dict[str, Any]] = [] + for item in payload.get("data", {}).get("result", []): + points = [ + [int(float(ts) * 1000), float(val)] + for ts, val in item.get("values", []) + ] + if points: + series.append({"metric": item.get("metric", {}), "points": points}) + return series + + +# ------------------------------------------------------------------ 查询 + + +@router.get("/health", summary="健康检查") +async def health(request: Request) -> dict[str, Any]: + controller = _controller(request) + snap = controller.snapshot() + return { + "status": "ok", + "control_loop_running": snap.running, + "mode": snap.mode, + "emergency": snap.emergency, + "consecutive_failures": snap.consecutive_failures, + } + + +@router.get("/status", summary="完整状态快照") +async def status(request: Request) -> dict[str, Any]: + return _controller(request).describe() + + +@router.get("/gpus", summary="GPU 指标") +async def gpus(request: Request) -> list[dict[str, Any]]: + return _controller(request).describe()["gpus"] + + +@router.get("/fans", summary="风扇位状态") +async def fans(request: Request) -> list[dict[str, Any]]: + return _controller(request).describe()["fans"] + + +@router.get("/curve", summary="当前控制曲线") +async def curve(request: Request) -> dict[str, Any]: + return _controller(request).describe()["curve"] + + +class SettingsPatch(BaseModel): + """运行时设置的部分更新。 + + 只传要改的键即可(**部分更新语义**)。 + """ + + control_enabled: bool | None = Field( + default=None, description="控制总开关。false = 完全不调档,风扇保持现状" + ) + control_interval: float | None = Field( + default=None, gt=0, le=3600, description="控制周期(秒)" + ) + curve: dict[str, Any] | None = Field( + default=None, + description="整条曲线 {points:[{temp,duty}], hysteresis, min_duty, max_duty}", + ) + emergency_temp: float | None = Field( + default=None, description="紧急散热触发温度(°C)" + ) + emergency_resume_temp: float | None = Field( + default=None, description="紧急散热解除温度(°C)" + ) + managed_gpus: list[str] | None = Field( + default=None, + description=( + "管控的 GPU UUID 列表(只控制这几张卡)。" + "空数组 = 全部管控(保守默认,新插的卡自动纳入)" + ), + ) + + +@router.get("/settings", summary="读取运行时设置") +async def get_settings(request: Request) -> dict[str, Any]: + """当前生效的设置。 + + 这些值来自 SQLite(配置文件只提供初始值)—— 也就是**界面上改过的就是权威值**。 + """ + return _controller(request).export_settings() + + +@router.patch("/settings", summary="修改运行时设置") +async def patch_settings( + payload: SettingsPatch, request: Request +) -> dict[str, Any]: + """部分更新:只传要改的键。 + + 改完立即生效(控制周期会在下一轮循环读到,曲线立即用于下一轮计算), + **不需要重启服务**,同时落库持久化。 + """ + controller = _controller(request) + store = getattr(request.app.state, "store", None) + + # 把 API 的扁平字段映射回内部的点号键 + patch: dict[str, Any] = {} + if payload.control_enabled is not None: + patch["control.enabled"] = payload.control_enabled + if payload.control_interval is not None: + patch["control.interval"] = payload.control_interval + if payload.curve is not None: + patch["curve"] = payload.curve + if payload.emergency_temp is not None: + patch["safety.emergency_temp"] = payload.emergency_temp + if payload.emergency_resume_temp is not None: + patch["safety.emergency_resume_temp"] = payload.emergency_resume_temp + if payload.managed_gpus is not None: + patch["control.managed_gpus"] = payload.managed_gpus + + if not patch: + raise HTTPException(status_code=400, detail="没有提供任何要修改的设置") + + try: + controller.apply_settings(patch) + except (ValueError, TypeError) as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + + # 管控范围变化会连带清除被移出卡的分配(controller 内处理), + # 这里把分配的最新状态一并落库,保证「设置 ↔ 分配」持久层也打通 + if payload.managed_gpus is not None and store is not None: + try: + store.save_assignments(controller.export_assignments()) + except Exception: # noqa: BLE001 + logger.exception("分配联动清除已生效,但持久化失败") + + if store is not None: + try: + store.save_settings(patch) + store.log( + "api_call", + "api", + "修改设置: " + + ", ".join(f"{k}={v!r}" for k, v in sorted(patch.items())), + ) + except Exception: # noqa: BLE001 - 持久化失败不该让设置回滚 + logger.exception("设置已生效,但持久化失败(重启后会回到旧值)") + + return {"ok": True, "settings": controller.export_settings()} + + +@router.get("/history", summary="历史趋势(代理 Prometheus query_range)") +async def history( + request: Request, + minutes: int = Query(default=30, ge=5, le=1440, description="回溯时长(分钟)"), +) -> dict[str, Any]: + """历史曲线。 + + **为什么由后端代理而不是前端直连 Prometheus**:避免跨域,也别把 + Prometheus 地址(以及它背后的整个监控栈)暴露到浏览器里。 + + 为什么历史要单独走 Prometheus:控制器只持有「此刻」的快照,历史时序 + 是 Prometheus 的主场。注意 Prometheus 是 30s 一次 scrape,所以曲线在 + 短时间内是阶梯状的 —— 这是数据源特性,不是画错了。 + + 未配置 ``sources.prometheus_url`` 时返回空序列(前端会提示), + 不算错误 —— 历史图是可选的。 + """ + config = request.app.state.config + src = config.sources + + if not src.prometheus_url: + return { + "minutes": minutes, + "series": [], + "note": "未配置 sources.prometheus_url,历史趋势不可用(实时数据不受影响)", + } + + end = time.time() + start = end - minutes * 60 + # step 不小于 30s(= Prometheus 的 scrape_interval), + # 再小只会在区间里拿到重复样本 + step = max(30.0, minutes * 60 / 400) + + targets: list[tuple[str, str, str, str]] = [] + if src.prometheus_gpu_instance: + targets.append( + ( + f'DCGM_FI_DEV_GPU_TEMP{{instance="{src.prometheus_gpu_instance}"}}', + "gpu_temp", + "°C", + "left", + ) + ) + if src.prometheus_fan_instance: + targets.append( + ( + f'ipmi_fan_speed_rpm{{instance="{src.prometheus_fan_instance}"}}', + "fan_rpm", + "RPM", + "right", + ) + ) + # CPU 温度 = **CPU 核温度 Tctl**,走 node_exporter —— 与「CPU 温度统一走 + # node_exporter」的决定一致。 + # ⚠️ 不用 BMC 的 `ipmi_temperature_celsius{name="CPU Temp"}` —— 那是主板 + # 传感器读数,语义上是「CPU 插槽附近的环境温度」,不是 CPU 核温度。 + # k10temp 在 hwmon 里的 chip 名是 PCI 路径形式(pci0000:00_0000:00:18_3), + # 所以必须靠 node_hwmon_sensor_label 做语义过滤,别硬编码 chip 名。 + if src.prometheus_node_instance: + targets.append( + ( + "node_hwmon_temp_celsius * on(chip, sensor) group_left(label) " + 'node_hwmon_sensor_label{label="Tctl", ' + f'instance="{src.prometheus_node_instance}"}}', + "cpu_temp", + "°C", + "left", + ) + ) + + series: list[dict[str, Any]] = [] + for promql, kind, unit, axis in targets: + try: + result = await asyncio.to_thread( + _query_range, src.prometheus_url, promql, start, end, step + ) + except (urllib.error.URLError, OSError, ValueError, RuntimeError) as exc: + logger.warning("查询历史失败(%s): %s", kind, exc) + continue + + for item in result: + metric = item["metric"] + if kind == "gpu_temp": + uuid = str(metric.get("UUID", "")) + label = f"GPU{metric.get('gpu', '?')} {uuid[4:12] or '?'}" + elif kind == "cpu_temp": + label = "CPU 温度" + else: + label = str(metric.get("name", "?")) + series.append( + { + "key": f"{kind}_{label}", + "label": label, + "unit": unit, + "axis": axis, + "points": item["points"], + } + ) + + return {"minutes": minutes, "series": series} + + +# ------------------------------------------------------------------ 调控 + + +@router.post("/mode", summary="切换控制模式") +async def set_mode(payload: ModeRequest, request: Request) -> dict[str, Any]: + controller = _controller(request) + try: + controller.set_mode(payload.mode) + except ValueError as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + return {"ok": True, "mode": controller.mode} + + +@router.post("/manual", summary="手动设定某个风扇位的占空比") +async def set_manual(payload: ManualDutyRequest, request: Request) -> dict[str, Any]: + controller = _controller(request) + try: + controller.set_manual_duty(payload.slot, payload.duty) + except RuntimeError as exc: + raise HTTPException(status_code=409, detail=str(exc)) from exc + except ValueError as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + return {"ok": True, "slot": payload.slot, "duty": payload.duty} + + +@router.post("/restore-auto", summary="立即把全部风扇位交回 BMC 自动控制") +async def restore_auto(request: Request) -> dict[str, Any]: + """"我要松手了" 按钮。 + + 这是**安全方向**的操作,任何时候都允许调用,不受当前模式限制 —— + 用户想紧急交还控制权的时候不该被任何状态检查拦住。 + """ + ipmi = getattr(request.app.state, "ipmi", None) + if ipmi is None: # pragma: no cover + raise HTTPException(status_code=503, detail="IPMI 客户端尚未初始化") + + result = await asyncio.to_thread(ipmi.restore_auto) + if result is None: + raise HTTPException(status_code=500, detail="回退失败,请手动检查风扇!") + + store = getattr(request.app.state, "store", None) + if store is not None: + store.log( + "ipmi_write", + "restore_auto", + f"手动触发「交回 BMC」 | {result.summary()}", + ok=result.ok, + ) + return {"ok": result.ok, "output": result.stdout.strip(), "rc": result.returncode} + + +@router.put("/assignments", summary="更新散热源与风扇位的分配") +async def update_assignments(payload: AssignmentsRequest, request: Request) -> dict[str, Any]: + """设置「哪个源用哪些风扇」—— **GPU 是主体,它挑自己的风扇接口**。 + + - ``kind=gpu`` 必须带 ``gpu_uuid``(**绝不用 index** —— 这台机器的两张卡 + 换过一次 PCI 槽位,index 和 pci_bus_id 都变过,只有 UUID/SN 稳定) + - 一个风扇位只能被一个源占用(冲突返回 400) + - ``slots`` 传空数组 = 该源不占用任何风扇(未分配的位程序一根线不碰) + - ``gpu:all`` 是「所有 GPU 最热」的合成源,用于还没摸清物理对应关系时 + 的安全配置 + """ + controller = _controller(request) + store = getattr(request.app.state, "store", None) + items = [a.model_dump() for a in payload.assignments] + + try: + controller.update_assignments(items) + except ValueError as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + + if store is not None: + try: + # 存控制器的**当前完整状态**,而不是请求体 —— 请求体可能只带了一部分。 + store.save_assignments(controller.export_assignments()) + store.log( + "api_call", + "api", + "更新分配: " + + "; ".join( + f"{a['key']}→{a['slots'] or '(无)'}" for a in items + ), + ) + except Exception: # noqa: BLE001 - 持久化失败不该让分配改动回滚 + logger.exception("分配已生效,但持久化失败(重启后会回到旧值)") + + return {"ok": True, "assignments": controller.describe()["assignments"]} + + +@router.get("/audit", summary="操作审计") +async def audit( + request: Request, + limit: int = Query(default=100, ge=1, le=1000, description="返回条数"), +) -> dict[str, Any]: + """最近的 IPMI 写入与 API 调用记录。 + + **这张表存在的意义**:2026-09-28 两个 GPU 风扇位从 3000 RPM 掉回 BMC + 自动档,uvicorn 日志被重启覆盖、IPMI raw 命令又不进 BMC SEL,**两边都查不出 + 是谁发的**。有了审计表,这类问题直接查库即可。 + """ + store = getattr(request.app.state, "store", None) + if store is None: # pragma: no cover + return {"records": []} + return {"records": store.recent_audit(limit)} + + +# ------------------------------------------------------------------ 实时推送 + + +@router.websocket("/ws") +async def websocket_status(websocket: WebSocket) -> None: + """按固定间隔推送状态快照。前端断了就静默收工,不刷日志。""" + await websocket.accept() + controller: FanController | None = getattr( + websocket.app.state, "controller", None + ) + if controller is None: # pragma: no cover + await websocket.close(code=1011) + return + + try: + while True: + await websocket.send_json(controller.describe()) + await asyncio.sleep(WS_PUSH_INTERVAL) + except WebSocketDisconnect: + logger.debug("前端 WebSocket 断开") + except Exception: # pragma: no cover - 网络抖动不该刷满日志 + logger.debug("WebSocket 推送异常,连接关闭", exc_info=True) diff --git a/app/config.py b/app/config.py new file mode 100644 index 0000000..b3e1112 --- /dev/null +++ b/app/config.py @@ -0,0 +1,161 @@ +"""应用配置模型与加载。 + +配置用 YAML(``app/config.yaml``),模型用 pydantic 校验 —— 配置写错时 +启动即报错,而不是跑起来才发现某个字段是字符串的 ``"50"``。 +""" + +from __future__ import annotations + +import logging +from pathlib import Path +from typing import Any, Literal + +import yaml +from pydantic import BaseModel, Field, field_validator + +logger = logging.getLogger(__name__) + +#: 项目根目录(``app/`` 的上一级) +PROJECT_ROOT = Path(__file__).resolve().parent.parent + +#: 默认配置文件位置 +DEFAULT_CONFIG_PATH = PROJECT_ROOT / "app" / "config.yaml" + + +class ServerConfig(BaseModel): + """HTTP 服务配置。""" + + host: str = "0.0.0.0" + port: int = Field(default=8765, ge=1, le=65535) + + +class ControlConfig(BaseModel): + """控制回路配置。""" + + enabled: bool = True + mode: Literal["auto", "manual"] = "auto" + #: 控制周期(秒) + interval: float = Field(default=15.0, gt=0) + #: 占空比变化小于此值时不重复下发,省得天天刷 BMC + min_write_delta: int = Field(default=3, ge=0, le=100) + #: 连续读取失败多少次后判定为「失明」 + max_consecutive_failures: int = Field(default=3, ge=1) + + +class SourcesConfig(BaseModel): + """数据源配置。""" + + # --- 实时读数:直连本机 exporter 的 /metrics(最低延迟)--- + dcgm_endpoint: str | None = "http://127.0.0.1:9400/metrics" + ipmi_exporter_endpoint: str | None = "http://127.0.0.1:9290/metrics" + #: node_exporter(需开 --collector.hwmon)—— 取 CPU 核心温度 Tctl/Tccd + node_exporter_endpoint: str | None = "http://127.0.0.1:9100/metrics" + ipmitool_binary: str = "ipmitool" + ipmi_timeout: float = Field(default=10.0, gt=0) + http_timeout: float = Field(default=5.0, gt=0) + #: 远程 BMC(``lanplus``)。留空则走本地 in-band(推荐) + ipmi_remote: dict[str, str] | None = None + + # --- 历史趋势(可选):只给 /api/history 用 --- + # 实时读数**不走**这里,因为 Prometheus 是 30s 快照(见 CLAUDE.md)。 + # 但历史曲线恰恰是 Prometheus 的主场,所以单独配一套。 + prometheus_url: str | None = None + #: GPU 温度在 Prometheus 里的 instance 标签,如 ``192.168.6.7:9400`` + prometheus_gpu_instance: str | None = None + #: CPU 温度(node_exporter 的 hwmon / k10temp)的 instance 标签,如 ``192.168.6.7:9100`` + prometheus_node_instance: str | None = None + #: 风扇转速的 instance 标签,如 ``192.168.6.7:9290`` + prometheus_fan_instance: str | None = None + + +class CurveConfig(BaseModel): + """温度-占空比曲线配置。""" + + points: list[dict[str, float]] = Field( + default_factory=lambda: [ + {"temp": 45, "duty": 40}, + {"temp": 55, "duty": 50}, + {"temp": 65, "duty": 65}, + {"temp": 75, "duty": 80}, + {"temp": 85, "duty": 100}, + ] + ) + #: 滞回带(°C)。降温方向必须跌出这个带宽才降档。 + hysteresis: float = Field(default=3.0, ge=0) + #: 占空比下限。BMC 本身可能拒绝低于约 20% 的值,这里再兜一层。 + min_duty: int = Field(default=30, ge=1, le=100) + max_duty: int = Field(default=100, ge=1, le=100) + + +class FanBinding(BaseModel): + """风扇位声明(**可选**)。 + + ⚠️ 这里**不包含分配关系**(哪张 GPU 用哪些风扇),也不需要把机器上的 + 风扇位全列出来 —— ``fan_slots`` 是**探测出来的**(配置声明的位 ∪ + ipmi_exporter 实测到有读数的位),用户在界面上从探测清单里勾选即可。 + + 配置里写位只有两个用途:① 提前占位(风扇还没转起来/没读数时也想让它 + 出现在清单里)② 限制定制范围。**留空(默认)= 完全交给自动探测**。 + + 为什么不把分配写进配置文件:那会造成**两个数据源**。用户在界面上改的 + 存进了数据库,回头改 config.yaml 却不生效(被数据库覆盖),非常容易把人 + 绕晕。单一来源,少一类 bug。(2026-09-28 超哥两次纠正过这个点。) + """ + + slot: str + #: 该风扇位的占空比上限(覆盖曲线的 max_duty);不填则用曲线自己的 + max_duty: int | None = Field(default=None, ge=1, le=100) + + +class SafetyConfig(BaseModel): + """安全护栏配置。""" + + #: 心跳文件路径。独立看门狗据此判断主进程是否还活着。 + heartbeat_path: str = "/run/gpu-fan-console/heartbeat" + #: 退出回退动作的超时(给得宽裕一点,回退是性命攸关的事) + restore_timeout: float = Field(default=15.0, gt=0) + #: 超过此温度立即拉满,不再等曲线 + emergency_temp: float = 85.0 + #: 温度回落到此值以下才解除紧急状态(同样需要滞回,否则会在临界点反复横跳) + emergency_resume_temp: float = 75.0 + + +class AppConfig(BaseModel): + """应用总配置。""" + + server: ServerConfig = Field(default_factory=ServerConfig) + control: ControlConfig = Field(default_factory=ControlConfig) + sources: SourcesConfig = Field(default_factory=SourcesConfig) + curve: CurveConfig = Field(default_factory=CurveConfig) + safety: SafetyConfig = Field(default_factory=SafetyConfig) + fans: list[FanBinding] = Field(default_factory=list) + + @field_validator("fans") + @classmethod + def _check_fan_slots(cls, value: list[FanBinding]) -> list[FanBinding]: + from .ipmi import FAN_SLOT_INDEX + + seen: set[str] = set() + for binding in value: + if binding.slot not in FAN_SLOT_INDEX: + raise ValueError( + f"未知风扇位 {binding.slot!r},合法取值: {sorted(FAN_SLOT_INDEX)}" + ) + if binding.slot in seen: + raise ValueError(f"风扇位 {binding.slot} 重复配置") + seen.add(binding.slot) + return value + + +def load_config(path: str | Path | None = None) -> AppConfig: + """从 YAML 加载配置。文件不存在时用默认值并给出提示。""" + config_path = Path(path) if path else DEFAULT_CONFIG_PATH + + if not config_path.exists(): + logger.warning("配置文件 %s 不存在,使用内置默认值", config_path) + return AppConfig() + + raw: Any = yaml.safe_load(config_path.read_text(encoding="utf-8")) or {} + config = AppConfig.model_validate(raw) + logger.info("已加载配置: %s", config_path) + return config diff --git a/app/config.yaml b/app/config.yaml new file mode 100644 index 0000000..ca46859 --- /dev/null +++ b/app/config.yaml @@ -0,0 +1,90 @@ +# GPU 风扇监控控制台 —— 配置 +# +# 改完这个文件需要重启服务生效: +# systemctl restart gpu-fan-console + +server: + host: "0.0.0.0" + port: 8765 + +control: + enabled: true + # auto = 按下面的 curve 自动调档(默认) + # manual = 只由界面手动指定占空比,控制器不插手 + mode: auto + # 控制周期(秒)。15s 对机箱风扇来说绰绰有余 —— 风道热惯性远大于这个量级。 + interval: 15 + # 占空比变化不足这个值就不重复下发,省得每轮都去打扰 BMC + min_write_delta: 3 + # 连续读不到温度多少次后开始大声报警 + max_consecutive_failures: 3 + +sources: + # GPU 指标:优先走本地 DCGM 端点(与 Prometheus 观测侧口径一致) + dcgm_endpoint: "http://127.0.0.1:9400/metrics" + # 风扇转速:优先走本地 ipmi_exporter + ipmi_exporter_endpoint: "http://127.0.0.1:9290/metrics" + ipmitool_binary: "ipmitool" + ipmi_timeout: 10 + http_timeout: 5 + + # 远程 BMC 兜底(宿主系统起不来时排障用)。默认走本地 in-band,不配这项。 + # ipmi_remote: + # host: "192.168.6.8" + # user: "admin" + # password: "admin" + + # --- 历史趋势(可选,只给 /api/history 用)--- + # + # 实时读数**不走**这里:Prometheus 是 30s 一次 scrape,拿到的永远是快照, + # 而上面两个 endpoint 是直连 exporter,读的才是当下值。 + # 但历史曲线恰恰是 Prometheus 的主场,所以单独配一套。 + # 不配的话界面上的趋势图会显示"暂无历史数据",其余功能不受影响。 + prometheus_url: "http://192.168.6.31:30091" + prometheus_gpu_instance: "192.168.6.7:9400" # GPU 温度 ← DCGM exporter + prometheus_node_instance: "192.168.6.7:9100" # CPU 温度 ← node_exporter + prometheus_fan_instance: "192.168.6.7:9290" # 风扇转速 ← ipmi_exporter + +curve: + # 滞回带(°C):降温方向必须跌出这个带宽才降档,防止温度在阈值附近 + # 抖动导致转速反复横跳("直升机效应")。 + hysteresis: 3 + # 占空比下限。EPYCD8 的 BMC 本身可能拒绝低于约 20% 的值,这里再兜一层。 + min_duty: 30 + max_duty: 100 + + # 温度 → 占空比。温度取该风扇位所绑定 GPU 的最大值。 + # Tesla T10 是被动散热卡,全靠机箱风扇吹;这张曲线偏保守, + # 实测过温降不下来就往下压阈值。 + points: + - { temp: 45, duty: 40 } + - { temp: 55, duty: 50 } + - { temp: 65, duty: 65 } + - { temp: 75, duty: 80 } + - { temp: 85, duty: 100 } + +# 风扇位声明(**可选,通常留空**) +# +# 「可分配的风扇位」是**探测出来的**:配置里写的位 ∪ ipmi_exporter 实测到 +# 有读数的位,在界面的分配面板里全部列出、由你勾选谁参与调控。 +# +# 这里写位只有两个特殊用途(一般用不上): +# ① 风扇还没读数时提前占位;② 限制定制范围。 +# +# ⚠️ 「哪张 GPU 用哪些风扇」的分配关系更不在这里 —— 那是运行时状态, +# 存 app/data/fan-console.db,界面上随改随生效。首次启动数据库为空 → +# 控制器不调档,界面弹分配向导,由你亲自给每张卡挑风扇(程序不替你猜)。 +fans: [] +# - slot: FRNT_FAN1 # 例:某个位没读数也想让它出现在清单里时才写 +# - slot: REAR_FAN2 + +safety: + # 心跳文件。独立的 systemd timer(fan-watchdog)会检查它的新鲜度 —— + # 这是 SIGKILL / 断电情况下唯一能救场的机制。 + heartbeat_path: "/run/gpu-fan-console/heartbeat" + # 退出回退动作的超时。给得宽裕些,回退是性命攸关的事。 + restore_timeout: 15 + # 超过这个温度立即拉满,不再等曲线 + emergency_temp: 85 + # 降回这个温度以下才解除紧急状态(同样要滞回,否则在临界点反复横跳) + emergency_resume_temp: 75 diff --git a/app/controller.py b/app/controller.py new file mode 100644 index 0000000..f255b0d --- /dev/null +++ b/app/controller.py @@ -0,0 +1,1074 @@ +"""控制回路。 + +每个周期做四件事:读 GPU 温度 → 查曲线算占空比 → 与上次下发值比较 → +必要时写 IPMI。看似简单,但「出错时怎么办」才是这个模块的重点: + +- **单次 tick 抛异常不能让回路死掉**。上游项目就是 ``while True`` 里没有 + 异常兜底,线程一崩进程还活着、但已经不再控风扇了 —— 这种静默失效 + 比直接崩溃危险得多。 +- **读不到温度时不要瞎猜**。既不盲目保持(可能正卡在低转速),也不盲目 + 拉满(可能把凉快的机器吹成噪音源)。正确做法是**什么都不做 + 报警**: + 保持现状,把决定权交给人和 BMC。 +- **降温方向要滞回**,否则温度在阈值附近抖动会让风扇转速反复横跳。 +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from dataclasses import dataclass, field +from typing import Any + +from .config import AppConfig +from .curve import CurveState, FanCurve, build_curve_from_config +from .ipmi import FAN_SLOT_INDEX, GPU_COOLING_SLOTS, IPMIClient +from .safety import SafetyGuard +from .sensors import ( + BoardTemperature, + BoardTemperatureReader, + CPUCoreTemperature, + CPUCoreTemperatureReader, + FanMetricsReader, + FanReading, + GPUMetric, + GPUMetricsReader, +) +from .store import Store + +logger = logging.getLogger(__name__) + + +@dataclass +class SourceAssignment: + """一个**散热源**分配到哪些风扇位(存在 SQLite,不来自配置文件)。 + + **主体是「源」,不是「风扇位」** —— 这条是 2026-09-28 超哥纠正的,我最初写反了: + + - ❌ 旧模型(我原来写的):风扇位 → 挑一路温度。要表达「FRNT_FAN1 给 GPU0 吹」, + 得滚到风扇那一行找下拉,是**反直觉**的。 + - ✅ 新模型:GPU → 挑它的风扇接口。用户脑子里的顺序是「**这张卡用哪个风扇吹**」, + 配置和界面就该按这个顺序组织。 + + 两者在「一个风扇位只被一个源占用」的前提下语义等价,但**表达顺序和界面形态 + 完全不同** —— 模型必须贴合用户的心智,不能只求数学等价。 + + 调控**完全由这份分配驱动**:被分配的位按对应源的温度调;没被分配的位 + 程序一根线都不碰(交回 BMC)。 + """ + + #: 源的唯一键:``gpu:`` 或 ``cpu`` + key: str + #: 源类型:``gpu`` = 某张具体的卡;``cpu`` = CPU 核温度 Tctl + kind: str = "gpu" + #: ``kind == "gpu"`` 时的 GPU UUID(**绝不用 index** —— 这机器换过 PCI 槽位) + gpu_uuid: str | None = None + #: 分配给这个源的风扇位(可多个) + slots: list[str] = field(default_factory=list) + + +@dataclass +class SlotStatus: + """单个风扇位的运行状态。""" + + slot: str + duty: int | None = None + temperature: float | None = None + curve_index: int | None = None + updated_ts: float | None = None + #: 本轮温度的来源说明(人话),如「GPU e49ed30f」/「所有卡最热」/「CPU Tctl」 + bound_detail: str = "" + #: 绑定的 GPU UUID(供前端展示绑定关系) + bound_uuids: list[str] = field(default_factory=list) + #: 温度源类型(gpu / cpu) + source: str = "gpu" + #: 占用这个风扇位的源的 key(空 = 没被分配,程序不接管) + owner_key: str = "" + #: 该位占空比上限(来自配置文件) + max_duty: int | None = None + + +@dataclass +class ControllerSnapshot: + """供 API 对外暴露的运行快照。""" + + running: bool = False + mode: str = "auto" + interval: float = 15.0 + last_tick_ts: float | None = None + last_tick_duration: float | None = None + consecutive_failures: int = 0 + last_error: str | None = None + emergency: bool = False + slots: dict[str, SlotStatus] = field(default_factory=dict) + gpus: list[GPUMetric] = field(default_factory=list) + fans: dict[str, FanReading] = field(default_factory=dict) + gpu_source: str = "" + fan_source: str = "" + #: BMC 板载温度(ipmi_exporter):MB / CPU / Card Side / DDR4_* + board_temps: dict[str, BoardTemperature] = field(default_factory=dict) + #: CPU 核心温度(node_exporter 的 hwmon):Tctl / Tccd* + cpu_temps: list[CPUCoreTemperature] = field(default_factory=list) + board_source: str = "" + cpu_source: str = "" + + +class FanController: + """GPU 温度 → 风扇占空比的闭环控制器。""" + + def __init__( + self, + config: AppConfig, + ipmi: IPMIClient, + guard: SafetyGuard, + gpu_reader: GPUMetricsReader, + fan_reader: FanMetricsReader, + curve: FanCurve, + board_reader: BoardTemperatureReader | None = None, + cpu_reader: CPUCoreTemperatureReader | None = None, + store: Store | None = None, + ) -> None: + self.config = config + self._ipmi = ipmi + self._guard = guard + self._gpu_reader = gpu_reader + self._fan_reader = fan_reader + self._curve = curve + self._board_reader = board_reader + self._cpu_reader = cpu_reader + self._store = store + + self._mode = config.control.mode + self._emergency = False + + # --- 管控范围 --- + # 「控制哪几张 GPU」。``None`` = 全部管控(保守默认:新插的卡自动纳入, + # 不然新卡没人吹会热死);非空 set = 只管这几个。设置页可改。 + self._managed_gpus: set[str] | None = None + + # --- 运行时设置 --- + # 下面这几项都能在界面上改、并持久化到 SQLite。配置文件只提供**初始值**, + # 之后一律以数据库为准(避免「改配置文件不生效」的双数据源困惑)。 + self._enabled = config.control.enabled + self._interval = config.control.interval + self._emergency_temp = config.safety.emergency_temp + self._emergency_resume = config.safety.emergency_resume_temp + self._curve_states: dict[str, CurveState] = { + b.slot: CurveState() for b in config.fans + } + #: 各风扇位的占空比上限(来自配置文件,可以按位覆盖曲线的 max_duty) + self._max_duty: dict[str, int | None] = { + b.slot: b.max_duty for b in config.fans + } + + # 分配关系是运行时状态,**权威来源是 SQLite**(见 app/store.py)。 + # 配置文件只回答一个问题:**这台机器上我要管哪几个风扇位**。 + # + # ⚠️ 这里初始是**空的**:程序不能替用户猜「哪个风扇给哪张卡散热」—— + # 猜错就是「凉的卡吹、热的卡不吹」,是要烧硬件的。所以库里没有分配时, + # 所有位都是**无主**状态 → 一个都不调,等用户在界面上完成初始化 + # (见 BindingWizard.vue)。 + self._assignments: dict[str, SourceAssignment] = {} + #: ``slot → 源的 key``。控制回路按「位」下发,靠这张索引找它属于谁。 + self._slot_owner: dict[str, str] = {} + self._bindings_configured = False + self._snapshot = ControllerSnapshot( + mode=self._mode, + interval=config.control.interval, + slots={b.slot: SlotStatus(slot=b.slot) for b in config.fans}, + ) + + # ------------------------------------------------------------ 属性 + + @property + def mode(self) -> str: + return self._mode + + @property + def interval(self) -> float: + """控制周期(秒)。运行时可改 —— 循环每轮重新读取,无需重启。""" + return self._interval + + @property + def enabled(self) -> bool: + return self._enabled + + def snapshot(self) -> ControllerSnapshot: + """当前运行快照的副本。""" + snap = self._snapshot + return ControllerSnapshot( + running=snap.running, + mode=self._mode, + interval=snap.interval, + last_tick_ts=snap.last_tick_ts, + last_tick_duration=snap.last_tick_duration, + consecutive_failures=snap.consecutive_failures, + last_error=snap.last_error, + emergency=self._emergency, + slots={ + slot: SlotStatus(**vars(status)) for slot, status in snap.slots.items() + }, + gpus=list(snap.gpus), + fans=dict(snap.fans), + gpu_source=self._gpu_reader.last_source, + fan_source=self._fan_reader.last_source, + board_temps=dict(snap.board_temps), + cpu_temps=list(snap.cpu_temps), + board_source=snap.board_source, + cpu_source=snap.cpu_source, + ) + + # ------------------------------------------------------------ 模式切换 + + def set_mode(self, mode: str) -> None: + """切换 ``auto`` / ``manual``。 + + 切回 ``auto`` 时重置曲线状态,避免拿着旧的档位索引做滞回判断。 + """ + if mode not in ("auto", "manual"): + raise ValueError(f"不支持的模式: {mode!r}(可选 auto / manual)") + if mode == self._mode: + return + logger.info("控制模式 %s → %s", self._mode, mode) + self._mode = mode + self._snapshot.mode = mode + for state in self._curve_states.values(): + state.reset() + + def update_assignments(self, assignments: list[dict[str, Any]]) -> None: + """更新「散热源 → 风扇位」的分配(内存态,持久化由调用方写 Store)。 + + 这是**全量覆盖**语义:传进来的就是完整的一份分配。 + + 校验规则: + + 1. ``kind`` 必须是 ``gpu`` / ``cpu`` 之一; + 2. ``kind == "gpu"`` 必须带 ``gpu_uuid``(**UUID,绝不用 index**); + 3. ``slots`` 里的每个风扇位必须是这台机器上**真实存在的位** + (配置文件声明的 + 探测到在转的,见 :meth:`_ensure_slot`); + 4. **一个风扇位只能被一个源占用** —— 冲突直接拒绝(整个更新不生效), + 否则「热的卡到底谁给它吹」就成了说不清的事; + 5. 同一个 ``key`` 出现两次 → 拒绝(key 是唯一键)。 + + 校验全部在**赋值之前** —— 中途抛错不会留下改了一半的状态。 + """ + if not isinstance(assignments, list): + raise ValueError("assignments 必须是数组") + + seen_keys: set[str] = set() + seen_slots: dict[str, str] = {} + + # ---- 第 1 遍:纯校验,不动任何状态 ---- + for item in assignments: + key = item.get("key") + kind = item.get("kind", "gpu") + if not isinstance(key, str) or not key: + raise ValueError("每个分配项都必须有非空 key") + if key in seen_keys: + raise ValueError(f"源 {key!r} 出现了两次") + seen_keys.add(key) + + if kind not in ("gpu", "cpu"): + raise ValueError(f"不支持的源类型: {kind!r}(可选 gpu / cpu)") + if kind == "gpu": + uuid = item.get("gpu_uuid") + if not isinstance(uuid, str) or not uuid: + raise ValueError(f"kind=gpu 的源({key!r})必须带 gpu_uuid") + if key != f"gpu:{uuid}": + raise ValueError(f"kind=gpu 的源 key 必须是 'gpu:'({key!r})") + # 打通「设置 ↔ 分配」:未纳入管控的卡不能**持有分配**, + # 否则用户配了也不生效、界面上却看不出来,两头糊涂。 + # (slots 为空的占位条目允许 —— 启动自愈后的形态就是这样) + if item.get("slots") and not self._is_managed(uuid): + raise ValueError( + f"GPU {uuid[4:12]} 未纳入管控(请在「设置 → 管控 GPU」里勾选)," + f"不能给它分配风扇" + ) + + slots = item.get("slots", []) + if not isinstance(slots, list): + raise ValueError(f"源 {key!r} 的 slots 必须是数组") + for raw_slot in slots: + slot = str(raw_slot) + # 校验基准 = 配置声明 ∪ 全量位表(懒建进运行时状态)。 + # ⚠️ 不能只看 self._snapshot.slots:启动时探测还没跑过(describe + # 尚未被调用),slots 可能是空的 —— 16:15 部署后曾因此把用户 + # 15:36 存的分配整份作废,风扇落回 BMC 失明档,GPU 飙到 86°C。 + if slot not in self._snapshot.slots and slot not in FAN_SLOT_INDEX: + raise ValueError(f"未纳入管控的风扇位: {slot!r}") + self._ensure_slot(slot) + if slot in seen_slots: + raise ValueError( + f"风扇位 {slot!r} 被两个源同时占用" + f"({seen_slots[slot]!r} 和 {key!r})—— 一个位只能给一个源" + ) + seen_slots[slot] = key + + # ---- 第 2 遍:校验全过了才真正赋值 ---- + self._assignments = { + item["key"]: SourceAssignment( + key=item["key"], + kind=item.get("kind", "gpu"), + gpu_uuid=item.get("gpu_uuid"), + slots=[str(s) for s in item.get("slots", [])], + ) + for item in assignments + } + self._reindex() + self._bindings_configured = True + + # 分配变化 → 档位索引全部作废(每个位的温度源可能换了) + for state in self._curve_states.values(): + state.reset() + + logger.info( + "分配更新: %s", + "; ".join( + f"{a.key} → [{', '.join(a.slots) or '未分配'}]" + for a in self._assignments.values() + ) + or "(空)", + ) + + def sanitize_stored_assignments( + self, assignments: list[dict[str, Any]] + ) -> list[dict[str, Any]]: + """启动恢复前的自愈:把与当前管控范围冲突的存量分配就地修正。 + + 「设置 ↔ 分配」在运行时是联动清除的(见 :meth:`apply_settings`), + 但如果上次持久化失败,库里可能残留「未管控的卡还带着分配」的脏数据。 + 启动恢复时**不能**因此把整份分配作废(16:56 事故的教训)—— + 这里只把冲突项的 ``slots`` 清空,其余原样保留。 + """ + for item in assignments: + if ( + item.get("kind") == "gpu" + and item.get("gpu_uuid") + and not self._is_managed(item["gpu_uuid"]) + and item.get("slots") + ): + logger.warning( + "启动自愈:GPU %s 未纳入管控,忽略其存量分配 %s", + item["gpu_uuid"][4:12], + item["slots"], + ) + item["slots"] = [] + return assignments + + def _reindex(self) -> None: + """重建 ``slot → 源 key`` 的索引(控制回路靠它按位找温度)。""" + self._slot_owner = { + slot: key + for key, a in self._assignments.items() + for slot in a.slots + } + # 把归属同步进每个位的运行状态(给前端展示「这个位被谁占着」) + for slot, status in self._snapshot.slots.items(): + owner = self._slot_owner.get(slot, "") + status.owner_key = owner + assignment = self._assignments.get(owner) if owner else None + status.source = assignment.kind if assignment else "gpu" + status.bound_uuids = ( + [assignment.gpu_uuid] if assignment and assignment.gpu_uuid else [] + ) + + def apply_settings(self, settings: dict[str, Any]) -> None: + """应用运行时设置(**部分更新**语义 —— 只改传进来的键)。 + + 支持的键: + + ================================ ========================================== + ``control.enabled`` 控制总开关。关闭 = 完全不调档,风扇保持现状 + ``control.interval`` 控制周期(秒),1~3600 + ``curve`` 整条曲线 ``{points, hysteresis, min_duty, max_duty}`` + ``safety.emergency_temp`` 紧急散热触发温度 + ``safety.emergency_resume_temp`` 紧急散热解除温度 + ================================ ========================================== + + Raises: + ValueError: 值非法(API 层会转成 400)。 + """ + if "control.enabled" in settings: + self._enabled = bool(settings["control.enabled"]) + logger.info("控制总开关 → %s", "开启" if self._enabled else "关闭") + + if "control.interval" in settings: + value = float(settings["control.interval"]) + if not 1 <= value <= 3600: + raise ValueError("控制周期需在 1~3600 秒之间") + self._interval = value + logger.info("控制周期 → %.1f 秒", value) + + if "curve" in settings: + raw = settings["curve"] or {} + points = raw.get("points") or [] + if not points: + raise ValueError("曲线至少要有一个折点") + self._curve = build_curve_from_config( + points, + hysteresis=float(raw.get("hysteresis", 3.0)), + min_duty=int(raw.get("min_duty", 20)), + max_duty=int(raw.get("max_duty", 100)), + ) + for state in self._curve_states.values(): + state.reset() + logger.info("控制曲线已更新(%d 个折点)", len(points)) + + if "safety.emergency_temp" in settings: + self._emergency_temp = float(settings["safety.emergency_temp"]) + if "safety.emergency_resume_temp" in settings: + self._emergency_resume = float(settings["safety.emergency_resume_temp"]) + + if "control.managed_gpus" in settings: + value = settings["control.managed_gpus"] + if not isinstance(value, list) or not all( + isinstance(u, str) for u in value + ): + raise ValueError("control.managed_gpus 必须是 UUID 字符串数组") + # 空数组 = 全部管控(保守默认);非空 = 只管列出的这几张 + self._managed_gpus = set(value) if value else None + logger.info( + "管控 GPU 范围 → %s", + f"指定 {len(value)} 张" if value else "全部", + ) + + # 打通「设置 ↔ 分配」:被移出管控的卡,其分配**立即停用并清除**。 + # 不留「配了但不生效」的暗状态 —— 重新勾选后需要重新分配。 + for a in self._assignments.values(): + if ( + a.kind == "gpu" + and a.gpu_uuid + and not self._is_managed(a.gpu_uuid) + and a.slots + ): + logger.info( + "GPU %s 已移出管控,其分配 %s 已停用清除", + a.gpu_uuid[4:12], + a.slots, + ) + a.slots = [] + self._reindex() + + def export_settings(self) -> dict[str, Any]: + """导出当前运行时设置(供持久化到 SQLite / 给前端展示)。""" + return { + "control.enabled": self._enabled, + "control.interval": self._interval, + "control.managed_gpus": sorted(self._managed_gpus or []), + "curve": self._curve.describe(), + "safety.emergency_temp": self._emergency_temp, + "safety.emergency_resume_temp": self._emergency_resume, + } + + def _is_managed(self, gpu_uuid: str) -> bool: + """该 GPU 是否被纳入管控(``None`` = 全部管控)。""" + return self._managed_gpus is None or gpu_uuid in self._managed_gpus + + def set_manual_duty(self, slot: str, duty: int | None) -> None: + """手动设定某个风扇位的占空比(仅在 ``manual`` 模式下允许)。 + + ``duty=None`` 表示把该位交回 BMC 自动。 + """ + if self._mode != "manual": + raise RuntimeError("当前不是手动模式,请先切换到 manual") + if slot not in self._snapshot.slots: + raise ValueError(f"未纳入管控的风扇位: {slot!r}") + + if not self._guard.engaged: + self._guard.engage() + + self._ipmi.apply({slot: duty}) + status = self._snapshot.slots[slot] + status.duty = duty + status.updated_ts = time.time() + + if self._store is not None: + self._store.log( + "ipmi_write", + "manual", + f"{slot}={'auto' if duty is None else f'{duty}%'}", + ) + + # ------------------------------------------------------------ 主循环 + + async def run(self) -> None: + """控制回路主循环(asyncio 后台任务)。""" + logger.info( + "控制回路启动 | 周期 %.1fs | 模式 %s | 管控风扇位 %s", + self.interval, + self._mode, + ", ".join(self._snapshot.slots), + ) + self._snapshot.running = True + try: + while True: + try: + # IPMI 与 HTTP 都是阻塞调用,丢到线程池里跑,别堵住事件循环 + await asyncio.to_thread(self.tick) + except asyncio.CancelledError: + raise + except Exception as exc: + # 单次失败绝不能终止回路 —— 这是上游最致命的坑 + logger.exception("控制周期出现未捕获异常(已吞掉,继续运行)") + self._snapshot.last_error = f"tick 异常: {exc}" + await asyncio.sleep(self.interval) + except asyncio.CancelledError: + logger.info("控制回路收到取消信号,退出") + raise + finally: + self._snapshot.running = False + + # ------------------------------------------------------------ 单次执行 + + def tick(self) -> None: + """执行一个控制周期。可单独调用,方便测试与排障。""" + started = time.monotonic() + + gpus = self._gpu_reader.read() + fans = self._fan_reader.read() + + # 温度是**附加信息**:读不到不影响控速(控速只看 GPU 温度), + # 所以放在最后,失败也不计入 consecutive_failures + if self._board_reader is not None: + self._snapshot.board_temps = self._board_reader.read() + self._snapshot.board_source = self._board_reader.last_source + if self._cpu_reader is not None: + self._snapshot.cpu_temps = self._cpu_reader.read() + self._snapshot.cpu_source = self._cpu_reader.last_source + + self._snapshot.gpus = gpus + self._snapshot.fans = fans + self._snapshot.gpu_source = self._gpu_reader.last_source + self._snapshot.fan_source = self._fan_reader.last_source + + if not gpus: + self._handle_blind("读不到任何 GPU 指标") + else: + self._snapshot.consecutive_failures = 0 + if self._mode == "auto" and self._enabled: + self._apply_curve(gpus) + elif self._mode == "manual": + logger.debug("手动模式,跳过自动调档") + elif not self._enabled: + logger.debug("控制总开关已关闭,跳过自动调档") + + self._snapshot.last_tick_ts = time.time() + self._snapshot.last_tick_duration = time.monotonic() - started + self._guard.beat(self._heartbeat_note()) + + # ------------------------------------------------------------ 曲线下发 + + def _apply_curve(self, gpus: list[GPUMetric]) -> None: + # 分配还没配置 → **不放任 BMC 失明档烤卡**。 + # 能走到这里说明总开关已开 —— 用户明确要求了自动控制,此时 BMC 的 + # 自动档对 GPU 完全失明(它读不到 GPU 温度),放着不管就是 16:56 + # 那种 86°C 险情。兜底:所有 GPU 散热位跟随最热卡跑同一条曲线 + # (只会多吹、不会漏吹)。用户在界面上完成分配后立即切换到精确分配。 + if not self._bindings_configured: + logger.warning( + "分配尚未配置 —— 兜底生效:GPU 散热位 %s 跟随最热卡跑曲线" + "(完成「风扇分配」后自动切换为精确控制)", + ", ".join(GPU_COOLING_SLOTS), + ) + self._run_fallback(gpus) + return + + # 没被任何源接管的卡:它的散热完全依赖 BMC 自动档,而 BMC 读不到 GPU + # 温度(这台机器的现实)。过热时必须把这件事说破,否则用户会以为 + # 「程序在管」—— 实际上一个风扇都没分给它。 + self._warn_orphan_gpus(gpus) + + # 紧急判定只看**被分配且纳入管控**的源 —— 只有它们驱动的位是程序能动的 + assigned_temps: list[float] = [ + g.temperature + for g in gpus + if g.temperature is not None + and f"gpu:{g.uuid}" in self._assignments + and self._is_managed(g.uuid) + ] + if any(a.kind == "cpu" for a in self._assignments.values()): + tctl = next( + (t for t in self._snapshot.cpu_temps if t.label == "Tctl"), None + ) + if tctl is not None and tctl.celsius is not None: + assigned_temps.append(tctl.celsius) + hottest = max(assigned_temps) if assigned_temps else None + + if hottest is not None: + self._update_emergency(hottest) + + updates: dict[str, int | None] = {} + pending: list[tuple[str, int]] = [] + + for assignment in self._assignments.values(): + if not assignment.slots: + continue + + # 未纳入管控的 GPU(设置页勾掉的):它的分配保留但**不驱动**, + # 风扇实际行为交回 BMC —— 用户明确说不管它,就不碰 + if ( + assignment.kind == "gpu" + and assignment.gpu_uuid + and not self._is_managed(assignment.gpu_uuid) + ): + for slot in assignment.slots: + status = self._snapshot.slots[slot] + status.bound_detail = "GPU 未纳入管控(设置页可改)" + status.temperature = None + continue + + temp, detail = self._resolve_temperature(assignment, gpus) + + # ⚠️ 拿不到温度(掉卡)→ **交回 BMC 自动**,而不是拉满狂转。 + # 掉卡的卡本身已经不发热了,为它狂转只是噪音;把位交回 BMC, + # 由它按机箱内其他传感器维持基本风道。卡恢复上线后自动重新接管。 + # (2026-09-28 17:20 超哥定:掉卡保持默认就好。) + if temp is None: + logger.warning( + "⚠️ %s 取不到温度(%s)—— 该风扇位交回 BMC 自动控制", + assignment.key, + detail, + ) + for slot in assignment.slots: + status = self._snapshot.slots[slot] + status.temperature = None + status.bound_detail = detail + # 只在当前不是自动时才写,避免每轮重复打扰 BMC + if status.duty is not None: + updates[slot] = None + pending.append((slot, 0)) + # 档位作废:恢复上线时按当时温度重新定档 + self._curve_states[slot].reset() + continue + + for slot in assignment.slots: + status = self._snapshot.slots[slot] + status.bound_detail = detail + status.temperature = temp + state = self._curve_states.setdefault(slot, CurveState()) + + if self._emergency: + duty = 100 + logger.warning( + "🚨 紧急状态:%s 直接拉满(%s)", + slot, + detail, + ) + else: + duty = self._curve.step(temp, state) + + max_duty = self._max_duty.get(slot) + if max_duty is not None: + duty = min(duty, max_duty) + + status.curve_index = state.index + + # 变化不够大就别打扰 BMC + if ( + status.duty is not None + and not self._emergency + and abs(duty - status.duty) < self.config.control.min_write_delta + ): + continue + + updates[slot] = duty + pending.append((slot, duty)) + + # 分配之外的位一根线都不碰 —— 这是「未分配 = 交回 BMC」的承诺 + if not updates: + return + + if not self._guard.engaged: + # 关键顺序:先武装护栏,再第一次写手动值。 + # 反过来的话,两次调用之间崩溃就没人负责回退了。 + self._guard.engage() + + result = self._ipmi.apply(updates) + detail = ", ".join(f"{slot}={duty}%" for slot, duty in pending) + + if result.ok: + now = time.time() + for slot, duty in pending: + self._snapshot.slots[slot].duty = duty + self._snapshot.slots[slot].updated_ts = now + if self._store is not None: + self._store.log("ipmi_write", "auto_curve", detail, ok=True) + else: + self._snapshot.last_error = f"下发失败: {result.summary()}" + if self._store is not None: + self._store.log( + "ipmi_write", + "auto_curve", + f"{detail} | {result.summary()}", + ok=False, + ) + + def _run_fallback(self, gpus: list[GPUMetric]) -> None: + """分配未配置时的安全兜底:所有 GPU 散热位跟随**最热卡**跑曲线。 + + 语义(超哥 16:56 明确要求):总开关开了,就该按曲线控制, + 而不是把风扇扔给读不到 GPU 温度的 BMC 自动档。 + """ + temps = [g.temperature for g in gpus if g.temperature is not None] + if not temps: + self._handle_blind("兜底模式:GPU 指标里没有任何温度读数") + return + + hottest = max(temps) + self._update_emergency(hottest) + + updates: dict[str, int | None] = {} + pending: list[tuple[str, int]] = [] + + for slot in GPU_COOLING_SLOTS: + if slot not in self._snapshot.slots and slot not in FAN_SLOT_INDEX: + continue + self._ensure_slot(slot) + status = self._snapshot.slots[slot] + status.owner_key = "__fallback__" + status.source = "gpu" + status.bound_uuids = [] + status.bound_detail = f"兜底 · 跟随最热卡({hottest:.0f}°C)" + status.temperature = hottest + + state = self._curve_states.setdefault(slot, CurveState()) + if self._emergency: + duty = 100 + logger.warning("🚨 紧急状态:%s 兜底直接拉满", slot) + else: + duty = self._curve.step(hottest, state) + + max_duty = self._max_duty.get(slot) + if max_duty is not None: + duty = min(duty, max_duty) + status.curve_index = state.index + + # 变化不够大就别打扰 BMC + if ( + status.duty is not None + and not self._emergency + and abs(duty - status.duty) < self.config.control.min_write_delta + ): + continue + + updates[slot] = duty + pending.append((slot, duty)) + + if not updates: + return + + if not self._guard.engaged: + self._guard.engage() + + result = self._ipmi.apply(updates) + detail = ", ".join(f"{slot}={duty}%" for slot, duty in pending) + + if result.ok: + now = time.time() + for slot, duty in pending: + st = self._snapshot.slots[slot] + st.duty = duty + st.updated_ts = now + if self._store is not None: + self._store.log("ipmi_write", "auto_fallback", detail, ok=True) + else: + self._snapshot.last_error = f"兜底下发失败: {result.summary()}" + if self._store is not None: + self._store.log( + "ipmi_write", + "auto_fallback", + f"{detail} | {result.summary()}", + ok=False, + ) + + def _warn_orphan_gpus(self, gpus: list[GPUMetric]) -> None: + """告警「没有任何风扇分给它的 GPU」—— 它的散热只剩 BMC 自动档兜底。 + + 只针对**纳入管控**的卡:用户在设置页主动排除的卡不唠叨。 + """ + for g in gpus: + if f"gpu:{g.uuid}" in self._assignments: + continue + if not self._is_managed(g.uuid): + continue + if g.temperature is not None and g.temperature >= self._emergency_resume: + logger.error( + "⚠️ GPU %s(%.0f°C)没有被分配任何风扇位 —— " + "它现在只有 BMC 自动档在散热,而 BMC 读不到 GPU 温度。" + "请到界面上给它分配风扇", + g.short_uuid, + g.temperature, + ) + + def export_assignments(self) -> list[dict[str, Any]]: + """导出当前分配(供持久化到 SQLite / 给前端展示)。""" + return [ + { + "key": a.key, + "kind": a.kind, + "gpu_uuid": a.gpu_uuid, + "slots": list(a.slots), + } + for a in self._assignments.values() + ] + + def _describe_assignments(self, gpus: list[GPUMetric]) -> list[dict[str, Any]]: + """给前端的分配视图。 + + **每张上报的 GPU 都是一个源**(哪怕还没分配风扇也列出来,slots 为空), + 外加一个 CPU 核温度源。 + + 配置过但当前离线的 GPU **照样列出**(标记 online=False)—— 否则它占着 + 的风扇位会在界面上凭空消失,用户会以为丢配置了。 + """ + by_key = {a.key: a for a in self._assignments.values()} + result: list[dict[str, Any]] = [] + + # ① 每张上报的 GPU + for g in gpus: + key = f"gpu:{g.uuid}" + stored = by_key.get(key) + result.append( + { + "key": key, + "kind": "gpu", + "gpu_uuid": g.uuid, + "label": f"GPU {g.short_uuid} · {g.model_name}", + "slots": list(stored.slots) if stored else [], + "temperature": g.temperature, + "online": True, + "managed": self._is_managed(g.uuid), + } + ) + + # ② 配置过但离线的 GPU(掉卡) + for key, stored in by_key.items(): + if stored.kind != "gpu" or not stored.gpu_uuid: + continue + if any(g.uuid == stored.gpu_uuid for g in gpus): + continue + result.append( + { + "key": key, + "kind": "gpu", + "gpu_uuid": stored.gpu_uuid, + "label": f"GPU {stored.gpu_uuid[4:12]} · 离线(掉卡?)", + "slots": list(stored.slots), + "temperature": None, + "online": False, + "managed": self._is_managed(stored.gpu_uuid), + } + ) + + # ③ 合成源:CPU 核温度 + cpu = by_key.get("cpu") + result.append( + { + "key": "cpu", + "kind": "cpu", + "gpu_uuid": None, + "label": "CPU 核温度(Tctl)", + "slots": list(cpu.slots) if cpu else [], + "temperature": None, + "online": True, + "managed": True, + } + ) + + return result + + def _resolve_temperature( + self, assignment: SourceAssignment, gpus: list[GPUMetric] + ) -> tuple[float | None, str]: + """解析某个**源**当前的温度。 + + Returns: + ``(温度, 人话说明)``。温度为 ``None`` 表示这一路取不到 —— + 调用方会据此走「保守拉满」的降级分支。 + """ + # ① CPU 源 + if assignment.kind == "cpu": + tctl = next( + (t for t in self._snapshot.cpu_temps if t.label == "Tctl"), None + ) + if tctl is None or tctl.celsius is None: + return None, "CPU 温度(Tctl)不可用" + return tctl.celsius, "CPU Tctl" + + # ② 具体某张 GPU + gpu = next((g for g in gpus if g.uuid == assignment.gpu_uuid), None) + if gpu is None or gpu.temperature is None: + return ( + None, + f"GPU {assignment.key[4:12]} 不在上报列表中(掉卡或未接入?)", + ) + return gpu.temperature, f"GPU {gpu.short_uuid}" + + def _update_emergency(self, hottest: float) -> None: + """维护紧急状态(进入和解除都要滞回,否则会在临界点反复横跳)。""" + if not self._emergency: + if hottest >= self._emergency_temp: + logger.error( + "🚨 进入紧急散热:%.1f°C ≥ %.1f°C", + hottest, + self._emergency_temp, + ) + self._emergency = True + elif hottest <= self._emergency_resume: + logger.warning( + "紧急散热解除:%.1f°C ≤ %.1f°C", + hottest, + self._emergency_resume, + ) + self._emergency = False + # 解除后重置曲线状态,重新从当前温度定档 + for state in self._curve_states.values(): + state.reset() + + # ------------------------------------------------------------ 失明处理 + + def _handle_blind(self, reason: str) -> None: + """读不到温度时的处理:**什么都不做 + 报警**。 + + 保持现状是这里唯一合理的选择——盲目拉满会把凉快的机器吹成噪音源, + 盲目按旧值调档则可能一路降速。把决定权留给人。 + """ + self._snapshot.consecutive_failures += 1 + self._snapshot.last_error = reason + failures = self._snapshot.consecutive_failures + + if failures == 1: + logger.warning("GPU 指标读取失败: %s", reason) + if failures >= self.config.control.max_consecutive_failures: + logger.error( + "⚠️ 已连续 %d 次读不到 GPU 温度 —— 保持当前占空比不动," + "请在界面上确认散热是否正常(当前管控位: %s)", + failures, + ", ".join( + f"{slot}={status.duty if status.duty is not None else 'auto'}" + for slot, status in self._snapshot.slots.items() + ), + ) + + # ------------------------------------------------------------ 杂项 + + def _heartbeat_note(self) -> str: + parts = [ + f"{slot}:{'auto' if s.duty is None else f'{s.duty}%'}" + for slot, s in self._snapshot.slots.items() + ] + return " ".join(parts) + + def _ensure_slot(self, slot: str) -> SlotStatus: + """确保某个风扇位的运行状态存在(探测到新位时**懒创建**)。 + + 为什么不靠配置文件:风扇位清单应该是**探测出来的**,不是部署时写死的 + (2026-09-28 超哥指出:界面固定两个位没道理,机器上明明有 4 个在转)。 + 现在的规则 —— ``fan_slots`` = 配置声明的位 ∪ ipmi_exporter 实测到 + 有读数的位,每轮 describe() 都会刷新,用户在界面上看到的就是 + 这台机器真实存在的全部风扇接口,勾谁控谁。 + """ + status = self._snapshot.slots.get(slot) + if status is None: + status = SlotStatus(slot=slot) + self._snapshot.slots[slot] = status + logger.info("探测到风扇位 %s(自动纳入可选清单)", slot) + self._curve_states.setdefault(slot, CurveState()) + return status + + def describe(self) -> dict[str, Any]: + """给前端的完整状态描述。""" + snap = self.snapshot() + # 风扇位**全部列出**(包括没有转速读数的)—— 控制器能写的位就是 + # FAN_SLOT_INDEX 覆盖的这些,少列一个用户就少一个可选项 + # (2026-09-28 超哥要求:没转速的也要展示)。 + for slot in FAN_SLOT_INDEX: + self._ensure_slot(slot) + return { + "running": snap.running, + "mode": snap.mode, + "interval": snap.interval, + "emergency": snap.emergency, + "last_tick_ts": snap.last_tick_ts, + "last_error": snap.last_error, + "consecutive_failures": snap.consecutive_failures, + "sources": {"gpu": snap.gpu_source, "fan": snap.fan_source}, + "gpus": [ + { + "uuid": g.uuid, + "short_uuid": g.short_uuid, + "index": g.index, + "pci_bus_id": g.pci_bus_id, + "model": g.model_name, + "temperature": g.temperature, + "power_watts": g.power_watts, + "utilization": g.utilization, + "memory_used_mib": g.memory_used_mib, + "memory_total_mib": g.memory_total_mib, + "memory_percent": g.memory_percent, + } + for g in snap.gpus + ], + "fans": [ + { + "slot": status.slot, + "duty": status.duty, + "temperature": status.temperature, + "curve_index": status.curve_index, + "updated_ts": status.updated_ts, + "source": status.source, + "owner_key": status.owner_key, + "bound_uuids": status.bound_uuids, + "bound_detail": status.bound_detail, + "rpm": ( + snap.fans[status.slot].rpm + if status.slot in snap.fans + else None + ), + } + for status in snap.slots.values() + ], + # 「散热源 → 风扇位」的分配(**主体是源**,界面上按 GPU 一行一行选风扇) + "assignments": self._describe_assignments(snap.gpus), + #: 全部风扇位 + 各自当前转速(没有读数的位 rpm=null,照样列出) + "fan_slots": [ + { + "slot": slot, + "rpm": (snap.fans[slot].rpm if slot in snap.fans else None), + } + for slot in sorted(self._snapshot.slots.keys()) + ], + #: false = 首次使用,前端应弹分配向导,且控制器不会调档 + "bindings_configured": self._bindings_configured, + "binding_options": { + "gpus": [ + { + "uuid": g.uuid, + "short_uuid": g.short_uuid, + "label": f"{g.short_uuid} · {g.model_name}", + } + for g in snap.gpus + ], + "cpu_label": "CPU 核温度(Tctl)", + }, + "curve": self._curve.describe(), + # 运行时设置(可在界面上改,落 SQLite) + "settings": self.export_settings(), + "ipmi_target": self._ipmi.target_state, + # 温度:CPU 核(node_exporter)+ 板载(BMC)。 + # ⚠️ GPU 温度不在这里 —— 它是控速的输入,放在 gpus[] 里。 + "temperatures": { + "cpu_cores": [ + {"label": t.label, "celsius": t.celsius} + for t in snap.cpu_temps + ], + "board": [ + {"name": t.name, "celsius": t.celsius, "state": t.state} + for t in sorted( + snap.board_temps.values(), + key=lambda x: (x.celsius is None, -(x.celsius or 0)), + ) + ], + "sources": {"cpu": snap.cpu_source, "board": snap.board_source}, + }, + } diff --git a/app/curve.py b/app/curve.py new file mode 100644 index 0000000..52edd74 --- /dev/null +++ b/app/curve.py @@ -0,0 +1,164 @@ +"""温度-占空比控制曲线(带滞回)。 + +**为什么不用 PID?** + +这个场景的被控对象是「机箱风扇 + 整条风道」,热惯性很大;而可用的调节手段 +只有 1%~100% 的整数占空比,分辨率相当粗。PWM 的粒度(1%)远大于温度噪声 +(±1°C),PID 在「粗粒度 + 大滞后」的组合下很容易震荡,调参成本还高。 + +分段曲线 + 滞回足够稳,而且有个更实际的好处:**一眼能看懂,随时能改**。 +出问题时你不需要去猜三个增益参数在干什么。 + +**滞回的必要性** + +温度在阈值附近抖动(比如 69.8 ↔ 70.2°C)时,没有滞回会让占空比在 +40% ↔ 60% 之间来回切换,风扇转速忽大忽小——就是俗称的「直升机效应」, +既吵又伤风扇。所以: + +- **升温方向立即生效**(散热是安全方向,不能延迟) +- **降温方向必须跌出滞回带才降档** +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Iterable, Sequence + +logger = logging.getLogger(__name__) + + +@dataclass(frozen=True) +class CurvePoint: + """曲线上的一个折点:温度达到 ``temp`` 时用 ``duty``。""" + + temp: float + duty: int + + def __post_init__(self) -> None: + if not 1 <= self.duty <= 100: + raise ValueError(f"占空比需在 1~100 之间,收到 {self.duty}") + + +@dataclass +class CurveState: + """曲线的运行状态。 + + 由调用方持有而非曲线自身持有 —— 曲线保持无状态,方便测试, + 也方便 API 层随时用不同温度试算而不污染运行状态。 + """ + + index: int | None = None + + def reset(self) -> None: + self.index = None + + +class FanCurve: + """分段温度-占空比曲线,带滞回防抖。 + + **阶梯语义**:温度必须**达到**某个折点才用那一档。例如折点 + ``[(70, 65), (80, 85)]`` 下,79°C 给的是 65% 而不是 85%。 + + 这是刻意选的保守约定。想要更平滑就多插几个折点,而不是改成线性插值 —— + 阶梯的行为可预测,排障时拿计算器一算就知道该给多少风。 + """ + + def __init__( + self, + points: Sequence[CurvePoint], + hysteresis: float = 3.0, + min_duty: int = 20, + max_duty: int = 100, + ) -> None: + if not points: + raise ValueError("曲线至少需要一个折点") + + ordered = sorted(points, key=lambda p: p.temp) + for prev, curr in zip(ordered, ordered[1:]): + if prev.temp == curr.temp: + raise ValueError(f"曲线折点温度重复: {curr.temp}°C") + + self._points: tuple[CurvePoint, ...] = tuple(ordered) + self.hysteresis = max(0.0, hysteresis) + self.min_duty = min_duty + self.max_duty = max_duty + + # ------------------------------------------------------------ 属性 + + @property + def points(self) -> tuple[CurvePoint, ...]: + return self._points + + # ------------------------------------------------------------ 计算 + + def _index_for(self, temp: float) -> int: + """温度 → 折点下标。低于最低折点时返回 0(即最低档)。""" + index = 0 + for i, point in enumerate(self._points): + if temp >= point.temp: + index = i + else: + break + return index + + def step(self, temp: float, state: CurveState) -> int: + """推进一次曲线,返回目标占空比(已按上下限钳制)。 + + Args: + temp: 当前温度(°C),多卡场景下传最大值。 + state: 可变状态,记录当前档位以实现滞回。 + """ + target = self._index_for(temp) + current = state.index + + if current is None: + state.index = target + logger.debug("曲线首次定档: %.1f°C → 第 %d 档 %d%%", temp, target, self._points[target].duty) + elif target > current: + # 升温:立即升档,不做延迟 + logger.debug( + "升温升档: %.1f°C → 第 %d 档 %d%%", + temp, target, self._points[target].duty, + ) + state.index = target + elif target < current: + # 降温:必须跌出滞回带才降档 + release_temp = self._points[current].temp - self.hysteresis + if temp <= release_temp: + logger.debug( + "降温降档: %.1f°C ≤ %.1f°C → 第 %d 档 %d%%", + temp, release_temp, target, self._points[target].duty, + ) + state.index = target + else: + logger.debug( + "降温但未跌出滞回带: %.1f°C > %.1f°C,保持第 %d 档", + temp, release_temp, current, + ) + + duty = self._points[state.index].duty + return max(self.min_duty, min(self.max_duty, duty)) + + def duty_at(self, temp: float) -> int: + """无状态试算:这个温度理论上该给多少占空比。 + + 给 API 层画曲线预览用,不影响运行状态。 + """ + duty = self._points[self._index_for(temp)].duty + return max(self.min_duty, min(self.max_duty, duty)) + + def describe(self) -> dict: + """序列化给前端展示。""" + return { + "points": [{"temp": p.temp, "duty": p.duty} for p in self._points], + "hysteresis": self.hysteresis, + "min_duty": self.min_duty, + "max_duty": self.max_duty, + } + + +def build_curve_from_config(raw_points: Iterable[dict], **kwargs) -> FanCurve: + """从配置里的一串 ``{"temp": .., "duty": ..}`` 构造曲线。""" + points = [CurvePoint(float(p["temp"]), int(p["duty"])) for p in raw_points] + return FanCurve(points, **kwargs) diff --git a/app/deploy/fan-watchdog.service b/app/deploy/fan-watchdog.service new file mode 100644 index 0000000..6dc43f6 --- /dev/null +++ b/app/deploy/fan-watchdog.service @@ -0,0 +1,8 @@ +[Unit] +Description=GPU Fan Console 心跳看门狗(单次执行) +# 独立于主服务:主服务卡死或被强杀时它才能救场 +After=multi-user.target + +[Service] +Type=oneshot +ExecStart=/opt/gpu-fan-console/app/deploy/fan-watchdog.sh diff --git a/app/deploy/fan-watchdog.sh b/app/deploy/fan-watchdog.sh new file mode 100644 index 0000000..52b5dbe --- /dev/null +++ b/app/deploy/fan-watchdog.sh @@ -0,0 +1,65 @@ +#!/bin/bash +# 心跳看门狗 —— 进程内捕获不到的退出场景的最后一道防线。 +# +# 背景:主进程在退出路径上会把风扇交回 BMC 自动控制(SafetyGuard 负责)。 +# 但有两种情况它做不到: +# 1. 被 SIGKILL 强杀(信号捕获不了) +# 2. 卡死在某个调用里,进程还在但控制回路已经停了 +# +# 本脚本由独立的 systemd timer 定期触发,发现心跳过期就强制回落。 +# 这个脚本故意写得"笨"——它只做一件事,且不依赖主程序的任何代码。 +# +# 部署:见同目录的 fan-watchdog.service / fan-watchdog.timer + +set -u + +HEARTBEAT="${HEARTBEAT:-/run/gpu-fan-console/heartbeat}" +# 心跳超时阈值(秒)。必须大于主进程控制周期 × 若干倍, +# 否则正常的短暂卡顿就会误触发。 +TIMEOUT="${TIMEOUT:-180}" +IPMITOOL="${IPMITOOL:-/usr/bin/ipmitool}" +SERVICE="${SERVICE:-gpu-fan-console.service}" + +# 全部风扇位交回 BMC 自动(8 字节必须写满,少一个字节 BMC 会静默忽略) +AUTO_PAYLOAD=(raw 0x3a 0x01 0x00 0x00 0x00 0x00 0x00 0x00 0x00 0x00) + +log() { + local msg="[fan-watchdog] $*" + echo "$(date '+%F %T') $msg" + command -v logger >/dev/null 2>&1 && logger -t fan-watchdog "$*" + return 0 +} + +# ---------------------------------------------------------------- 主逻辑 + +if [ ! -f "$HEARTBEAT" ]; then + # 没有心跳文件 = 主进程从未启动过,或者已经正常停止(正常停止会清理心跳 + # 并且已经回落过)。两种情况都不需要干预。 + exit 0 +fi + +mtime=$(stat -c %Y "$HEARTBEAT" 2>/dev/null || echo 0) +now=$(date +%s) +age=$(( now - mtime )) + +if [ "$age" -le "$TIMEOUT" ]; then + exit 0 # 心跳新鲜,一切正常 +fi + +log "⚠️ 心跳已过期 ${age}s(阈值 ${TIMEOUT}s)—— 主进程可能已卡死或被强杀" + +if systemctl is-active --quiet "$SERVICE" 2>/dev/null; then + log "服务仍显示 active 但心跳过期,判定为卡死,执行强制回落" +else + log "服务已非 active,执行强制回落" +fi + +if "$IPMITOOL" "${AUTO_PAYLOAD[@]}" >/dev/null 2>&1; then + log "✅ 已强制回落 BMC 自动控制" + # 清掉过期心跳,避免下一次触发时重复告警 + rm -f "$HEARTBEAT" + exit 0 +else + log "❌ 强制回落失败!请立即手动检查风扇状态(ipmitool sdr type fan)" + exit 1 +fi diff --git a/app/deploy/fan-watchdog.timer b/app/deploy/fan-watchdog.timer new file mode 100644 index 0000000..4c020a7 --- /dev/null +++ b/app/deploy/fan-watchdog.timer @@ -0,0 +1,12 @@ +[Unit] +Description=定时触发 GPU 风扇看门狗心跳检查 + +[Timer] +# 开机 3 分钟后开始第一次检查(给主服务留出启动时间) +OnBootSec=3min +# 之后每 2 分钟一次 +OnUnitActiveSec=2min +AccuracySec=10s + +[Install] +WantedBy=timers.target diff --git a/app/deploy/gpu-fan-console.service b/app/deploy/gpu-fan-console.service new file mode 100644 index 0000000..d786392 --- /dev/null +++ b/app/deploy/gpu-fan-console.service @@ -0,0 +1,28 @@ +[Unit] +Description=GPU Fan Console (pve02 GPU 温度联动风扇控制台) +Documentation=https://github.com/chennest/python-ipmitool +# docker 起来是为了让 dcgm-exporter / ipmi_exporter 先就绪; +# 但即使它们没起来,本服务也能跑(会降级到 nvidia-smi 兜底)。 +After=network-online.target docker.service +Wants=network-online.target + +[Service] +Type=simple +# 本地 in-band 读写 /dev/ipmi0 需要 root +User=root +WorkingDirectory=/opt/gpu-fan-console +ExecStart=/opt/gpu-fan-console/.venv/bin/python -m uvicorn app.main:app --host 0.0.0.0 --port 8765 + +Restart=always +RestartSec=5 + +# 停止时给足时间执行「回退 BMC 自动控制」——这是本服务最重要的退出动作 +TimeoutStopSec=30 +KillSignal=SIGTERM + +StandardOutput=journal +StandardError=journal +SyslogIdentifier=gpu-fan-console + +[Install] +WantedBy=multi-user.target diff --git a/app/diagnose.py b/app/diagnose.py new file mode 100644 index 0000000..3c6eeb2 --- /dev/null +++ b/app/diagnose.py @@ -0,0 +1,268 @@ +"""只读诊断工具 —— 不碰风扇,只报告读数。 + +**安全设计:本模块永远不会调用 ``apply()``,只读不写。** + +用途: + +1. 首次部署时验证数据源是否通(DCGM / ipmi_exporter / ipmitool) +2. 排障时确认「读数到底对不对」 +3. 调曲线之前预估每个风扇位会拿到什么占空比 + +用法:: + + python -m app.diagnose + python -m app.diagnose --json # 给脚本消费 + python -m app.diagnose --config /path/to/config.yaml +""" + +from __future__ import annotations + +import argparse +import json +import logging +import sys +from typing import Any + +from .config import AppConfig, load_config +from .curve import CurveState, build_curve_from_config +from .ipmi import IPMIClient +from .sensors import BoardTemperatureReader, FanMetricsReader, GPUMetricsReader + + +def collect(config: AppConfig) -> dict[str, Any]: + """采集一次完整读数(只读)。""" + gpu_reader = GPUMetricsReader( + dcgm_endpoint=config.sources.dcgm_endpoint, + timeout=config.sources.http_timeout, + ) + fan_reader = FanMetricsReader( + exporter_endpoint=config.sources.ipmi_exporter_endpoint, + ipmi_binary=config.sources.ipmitool_binary, + timeout=config.sources.http_timeout, + ) + curve = build_curve_from_config( + config.curve.points, + hysteresis=config.curve.hysteresis, + min_duty=config.curve.min_duty, + max_duty=config.curve.max_duty, + ) + ipmi = IPMIClient( + binary=config.sources.ipmitool_binary, + remote=config.sources.ipmi_remote, + timeout=config.sources.ipmi_timeout, + dry_run=True, + ) + + gpus = gpu_reader.read() + fans = fan_reader.read() + board_reader = BoardTemperatureReader( + exporter_endpoint=config.sources.ipmi_exporter_endpoint, + ipmi_binary=config.sources.ipmitool_binary, + timeout=config.sources.http_timeout, + ) + board_temps = board_reader.read() + + gpu_payload = [ + { + "index": g.index, + "uuid": g.uuid, + "short_uuid": g.short_uuid, + "pci_bus_id": g.pci_bus_id, + "model": g.model_name, + "temperature_c": g.temperature, + "power_watts": g.power_watts, + "utilization_pct": g.utilization, + "memory_used_mib": g.memory_used_mib, + "memory_total_mib": g.memory_total_mib, + "memory_pct": g.memory_percent, + } + for g in gpus + ] + + all_temps = [g.temperature for g in gpus if g.temperature is not None] + hottest = max(all_temps) if all_temps else None + + decisions = [] + for binding in config.fans: + if binding.gpu_uuids: + temps = [ + g.temperature + for g in gpus + if g.uuid in binding.gpu_uuids and g.temperature is not None + ] + scope = f"绑定 {len(binding.gpu_uuids)} 张卡" + else: + temps = all_temps + scope = "跟随所有卡(未绑定)" + + if not temps: + decisions.append( + { + "slot": binding.slot, + "scope": scope, + "temperature_c": None, + "target_duty": None, + "note": "无温度读数,实际运行时本轮会跳过", + } + ) + continue + + temp = max(temps) + duty = curve.step(temp, CurveState()) + if binding.max_duty is not None: + duty = min(duty, binding.max_duty) + decisions.append( + { + "slot": binding.slot, + "scope": scope, + "temperature_c": temp, + "target_duty": duty, + "note": "", + } + ) + + return { + "sources": { + "gpu": gpu_reader.last_source or "(未取到)", + "fan": fan_reader.last_source or "(未取到)", + "dcgm_endpoint": config.sources.dcgm_endpoint, + "ipmi_exporter_endpoint": config.sources.ipmi_exporter_endpoint, + }, + "gpus": gpu_payload, + "fans": [ + {"slot": r.slot, "rpm": r.rpm, "state": r.state, "source": r.source} + for r in fans.values() + ], + "board_temperatures": [ + {"name": t.name, "celsius": t.celsius, "state": t.state} + for t in board_temps.values() + ], + "hottest_gpu_c": hottest, + "emergency": bool( + hottest is not None and hottest >= config.safety.emergency_temp + ), + "curve": curve.describe(), + "decisions": decisions, + "ipmi_would_send": { + "target_state": ipmi.target_state, + "note": "dry-run,本工具不会真的下发", + }, + } + + +def render(data: dict[str, Any]) -> str: + """人类可读输出。""" + lines: list[str] = [] + add = lines.append + + add("=" * 68) + add("GPU 风扇控制台 · 只读诊断(不会下发任何命令)") + add("=" * 68) + + src = data["sources"] + add("") + add(f"数据源 GPU: {src['gpu']} ← {src['dcgm_endpoint']}") + add(f" 风扇: {src['fan']} ← {src['ipmi_exporter_endpoint']}") + + add("") + add(f"GPU({len(data['gpus'])} 张)") + if not data["gpus"]: + add(" ⚠️ 一张都没读到 —— 检查 DCGM exporter 是否在跑") + for g in data["gpus"]: + temp = f"{g['temperature_c']}°C" if g["temperature_c"] is not None else "n/a" + power = f"{g['power_watts']:.1f}W" if g["power_watts"] is not None else "n/a" + util = ( + f"{g['utilization_pct']}%" + if g["utilization_pct"] is not None + else "n/a" + ) + mem = ( + f"{g['memory_used_mib']:.0f}/{g['memory_total_mib']:.0f} MiB" + if g["memory_total_mib"] + else "n/a" + ) + add( + f" [{g['index']}] {g['short_uuid']} {g['model']} " + f"{g['pci_bus_id'].replace('00000000:', '')}" + ) + add(f" 温度 {temp} | 功耗 {power} | 利用率 {util} | 显存 {mem}") + + add("") + add(f"风扇({len(data['fans'])} 个位)") + if not data["fans"]: + add(" ⚠️ 一个都没读到 —— 检查 ipmi_exporter 是否在跑") + for f in data["fans"]: + rpm = f"{f['rpm']:.0f} RPM" if f["rpm"] is not None else "no reading" + add(f" {f['slot']:<12} {rpm:>12} {f['state']}") + + if data.get("board_temperatures"): + add("") + add("BMC 板载温度(⚠️ 都不是 GPU 温度 —— BMC 读不到 GPU)") + for t in sorted( + data["board_temperatures"], + key=lambda x: (x["celsius"] is None, -(x["celsius"] or 0)), + ): + celsius = f"{t['celsius']:.0f}°C" if t["celsius"] is not None else "n/a" + add(f" {t['name']:<18} {celsius:>7} {t['state']}") + + add("") + hottest = data["hottest_gpu_c"] + add(f"最热 GPU: {hottest}°C" if hottest is not None else "最热 GPU: n/a") + if data["emergency"]: + add("🚨 已超过紧急阈值 —— 实际运行时会直接拉满 100%") + + add("") + add("曲线") + pts = " ".join(f"{p['temp']}°C→{p['duty']}%" for p in data["curve"]["points"]) + add(f" {pts}") + add( + f" 滞回 {data['curve']['hysteresis']}°C | 下限 {data['curve']['min_duty']}%" + f" | 上限 {data['curve']['max_duty']}%" + ) + + add("") + add("按当前温度,曲线会给出的目标占空比") + for d in data["decisions"]: + if d["target_duty"] is None: + add(f" {d['slot']:<12} —— {d['scope']}|{d['note']}") + else: + add( + f" {d['slot']:<12} {d['target_duty']:>3}% " + f"({d['scope']},{d['temperature_c']}°C)" + ) + + add("") + add("=" * 68) + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="GPU 风扇控制台只读诊断(不写 IPMI)" + ) + parser.add_argument("--config", default=None, help="配置文件路径") + parser.add_argument("--json", action="store_true", help="输出 JSON") + parser.add_argument("-v", "--verbose", action="store_true", help="显示调试日志") + args = parser.parse_args(argv) + + logging.basicConfig( + level=logging.DEBUG if args.verbose else logging.WARNING, + format="%(asctime)s %(levelname)-7s [%(name)s] %(message)s", + ) + + try: + config = load_config(args.config) + except Exception as exc: + print(f"❌ 配置加载失败: {exc}", file=sys.stderr) + return 2 + + data = collect(config) + if args.json: + print(json.dumps(data, ensure_ascii=False, indent=2)) + else: + print(render(data)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/app/ipmi.py b/app/ipmi.py new file mode 100644 index 0000000..eabbdcf --- /dev/null +++ b/app/ipmi.py @@ -0,0 +1,339 @@ +"""IPMI 风扇控制底层封装。 + +目标平台:ASRock Rack EPYCD8(BMC 固件 2.20),命令族 ``raw 0x3a 0x01``。 + +⚠️ 三条硬约束 —— 动这个文件之前请先读完: + +1. **必须一次写满 8 个字节。** + 少写字节时 BMC 不报错、返回码仍为 0,但目标风扇转速纹丝不动。 + (2026-09-17 实测:误给 7 字节,一度误判「该风扇位不可控」,补齐后立即生效。) + +2. **单字节取值语义** + - ``0x00`` = 交回 BMC 自动控制(硬件 Smart Fan 温度-占空比表生效) + - ``0x01``~``0x64`` = 手动占空比百分比。字节的**十进制值**即百分比: + ``0x14``=20%、``0x32``=50%、``0x64``=100% + - BMC 可能拒绝低于约 20% 的占空比 + +3. **手动值不持久化** + BMC 重启或整机断电后自动回到 BMC 自动策略;CPU 温度达到临界阈值时 + BMC 会强行覆盖手动值。这是热保护,**不可对抗,也不应尝试对抗**。 + +8 字节位映射(b2 为保留位,恒 ``0x00``):: + + b1 CPU1_FAN1 + b2 --(保留) + b3 REAR_FAN1 + b4 REAR_FAN2 ← pve02 用于 Tesla T10 散热 + b5 FRNT_FAN1 ← pve02 用于 Tesla T10 散热 + b6 FRNT_FAN2 (未接风扇) + b7 FRNT_FAN3 (未接风扇) + b8 FRNT_FAN4 (未接风扇) + +相对上游 python-ipmitool 的两处关键改进: + +- 所有子进程调用**强制带超时**(上游用 ``p.stdout.read()`` 无超时,ipmitool + 一旦卡住会静默挂死整个线程,进程活着但不干活 —— 最难排查的失效形态) +- 在「必须全量写」的前提下,本地维护目标状态,从而能精确控制单个风扇位, + 不会误踩其它位 +""" + +from __future__ import annotations + +import logging +import subprocess +import time +from dataclasses import dataclass +from typing import Iterable, Mapping + +logger = logging.getLogger(__name__) + +# --------------------------------------------------------------------- 常量 + +#: 风扇位名称 → 8 字节 payload 中的下标(0-based) +FAN_SLOT_INDEX: dict[str, int] = { + "CPU1_FAN1": 0, + "REAR_FAN1": 2, + "REAR_FAN2": 3, + "FRNT_FAN1": 4, + "FRNT_FAN2": 5, + "FRNT_FAN3": 6, + "FRNT_FAN4": 7, +} + +#: payload 中保留位(b2)的下标 +RESERVED_INDEX: int = 1 + +#: payload 长度。少一个字节 BMC 会静默忽略整条命令,所以这是个硬性校验点。 +PAYLOAD_LEN: int = 8 + +#: 手动占空比取值范围(0 被保留用于表示「自动」) +MIN_DUTY: int = 1 +MAX_DUTY: int = 100 + +#: 语义常量:交回 BMC 自动控制 +AUTO: None = None + +#: 控制命令的 opcode 前缀 +CMD_PREFIX: tuple[str, ...] = ("raw", "0x3a", "0x01") + +#: 需要管控的风扇位(供上层循环使用) +GPU_COOLING_SLOTS: tuple[str, ...] = ("FRNT_FAN1", "REAR_FAN2") + + +# --------------------------------------------------------------------- 异常 + + +class IPMIError(RuntimeError): + """IPMI 操作失败基类。""" + + +class IPMITimeoutError(IPMIError): + """ipmitool 调用超时。""" + + +class IPMINotFoundError(IPMIError): + """找不到 ipmitool 可执行文件。""" + + +# --------------------------------------------------------------------- 结果 + + +@dataclass(frozen=True) +class CmdResult: + """一次 ipmitool 调用的结果。""" + + args: tuple[str, ...] + returncode: int + stdout: str + stderr: str + duration: float + + @property + def ok(self) -> bool: + return self.returncode == 0 + + def summary(self) -> str: + return ( + f"rc={self.returncode} {self.duration:.2f}s " + f"stdout={self.stdout.strip()!r} stderr={self.stderr.strip()!r}" + ) + + +# --------------------------------------------------------------------- 编解码 + + +def _fmt_byte(value: int) -> str: + """把 0~255 的字节值格式化成 ipmitool 需要的十六进制字面量。""" + return f"0x{value & 0xFF:02x}" + + +def encode_duty(duty: int | None) -> int: + """占空比 → 单字节值。 + + ``None`` 表示交回 BMC 自动(编码为 ``0x00``);``1~100`` 表示手动百分比, + 字节的十进制值就是百分比本身(``50`` → ``0x32``)。 + """ + if duty is None: + return 0x00 + if isinstance(duty, bool) or not isinstance(duty, int): + raise TypeError(f"占空比必须是 int 或 None,收到 {type(duty).__name__}") + if not MIN_DUTY <= duty <= MAX_DUTY: + raise ValueError(f"占空比需在 {MIN_DUTY}~{MAX_DUTY} 之间,收到 {duty}") + return duty + + +# --------------------------------------------------------------------- 客户端 + + +class IPMIClient: + """ipmitool 封装。 + + 默认走**本地 in-band**(``/dev/ipmi0``,需要 root),这是 pve02 上的推荐用法: + 链路最短、无网络依赖。也支持 ``lanplus`` 远程模式指向 BMC 独立地址 + (``192.168.6.8``),用于宿主系统起不来时的带外兜底。 + """ + + def __init__( + self, + binary: str = "ipmitool", + remote: Mapping[str, str] | None = None, + timeout: float = 10.0, + dry_run: bool = False, + ) -> None: + self.binary = binary + self.remote = dict(remote) if remote else None + self.timeout = timeout + self.dry_run = dry_run + # 目标状态:风扇位 → 占空比(None = BMC 自动)。 + # BMC 要求 8 字节全量写,所以必须靠这份状态拼出完整 payload, + # 否则「只想改一个位」就会把其它位一起踩成 0x00。 + self._target: dict[str, int | None] = {slot: AUTO for slot in FAN_SLOT_INDEX} + + # ------------------------------------------------------------ 内部工具 + + def _base_args(self) -> list[str]: + if not self.remote: + return [] + return [ + "-I", + "lanplus", + "-H", + str(self.remote.get("host", "")), + "-U", + str(self.remote.get("user", "admin")), + "-P", + str(self.remote.get("password", "")), + ] + + @staticmethod + def _redact(cmd: Iterable[str]) -> list[str]: + """日志脱敏:别把 BMC 密码打进日志文件。""" + out: list[str] = [] + mask_next = False + for token in cmd: + if mask_next: + out.append("***") + mask_next = False + continue + out.append(token) + if token == "-P": + mask_next = True + return out + + # ------------------------------------------------------------ 执行 + + def run(self, args: Iterable[str], timeout: float | None = None) -> CmdResult: + """执行一次 ipmitool 调用。超时抛 :class:`IPMITimeoutError`。""" + cmd = [self.binary, *self._base_args(), *args] + effective_timeout = self.timeout if timeout is None else timeout + started = time.monotonic() + + logger.debug("执行: %s", " ".join(self._redact(cmd))) + + try: + proc = subprocess.run( + cmd, + capture_output=True, + text=True, + timeout=effective_timeout, + check=False, + ) + except subprocess.TimeoutExpired as exc: + logger.error( + "ipmitool 超时(%.1fs 未返回): %s", + effective_timeout, + " ".join(self._redact(cmd)), + ) + raise IPMITimeoutError( + f"ipmitool 调用超过 {effective_timeout}s 未返回" + ) from exc + except FileNotFoundError as exc: + raise IPMINotFoundError( + f"找不到 ipmitool 可执行文件: {self.binary!r},请先安装 ipmitool" + ) from exc + + result = CmdResult( + args=tuple(cmd), + returncode=proc.returncode, + stdout=proc.stdout or "", + stderr=proc.stderr or "", + duration=time.monotonic() - started, + ) + if not result.ok: + logger.warning("ipmitool 返回非零: %s", result.summary()) + return result + + # ------------------------------------------------------------ 目标状态 + + @property + def target_state(self) -> dict[str, int | None]: + """当前目标状态的副本(风扇位 → 占空比 / None)。""" + return dict(self._target) + + def describe_target(self) -> str: + parts = [ + f"{slot}={'auto' if duty is None else f'{duty}%'}" + for slot, duty in self._target.items() + ] + return " ".join(parts) + + def build_payload(self, updates: Mapping[str, int | None] | None = None) -> list[int]: + """合并目标状态并编码成 8 字节 payload。 + + Args: + updates: 风扇位 → 占空比。``None`` 表示该位交回 BMC 自动。 + 未出现的位保持上一次设定的值。 + + Raises: + ValueError: 出现未知风扇位,或占空比超出取值范围。 + """ + if updates: + unknown = set(updates) - set(FAN_SLOT_INDEX) + if unknown: + raise ValueError( + f"未知风扇位 {sorted(unknown)},合法取值: {sorted(FAN_SLOT_INDEX)}" + ) + self._target.update(updates) + + payload = [0x00] * PAYLOAD_LEN + for slot, duty in self._target.items(): + payload[FAN_SLOT_INDEX[slot]] = encode_duty(duty) + # 保留位恒 0x00(_target 里没有它,这里显式兜一层) + payload[RESERVED_INDEX] = 0x00 + + if len(payload) != PAYLOAD_LEN: # pragma: no cover - 防御性断言 + raise AssertionError(f"payload 长度异常: {len(payload)}") + return payload + + # ------------------------------------------------------------ 下发 + + def apply( + self, + updates: Mapping[str, int | None] | None = None, + timeout: float | None = None, + ) -> CmdResult: + """更新目标状态并下发(8 字节全量写)。 + + Args: + updates: 风扇位 → 占空比,``None`` 表示交回 BMC 自动。 + 未出现的位保持上一次设定的值(首次为自动)。 + timeout: 覆盖默认超时,给退出路径上的回退调用留更宽裕的时间。 + + Returns: + 一次 ipmitool 调用的结果。 + """ + payload = self.build_payload(updates) + args = [*CMD_PREFIX, *(_fmt_byte(b) for b in payload)] + + if self.dry_run: + logger.info("[dry-run] 不下发,仅演示: ipmitool %s", " ".join(args)) + return CmdResult( + args=tuple(args), returncode=0, stdout="", stderr="", duration=0.0 + ) + + result = self.run(args, timeout=timeout) + if result.ok: + logger.info("风扇占空比已下发 | %s", self.describe_target()) + return result + + def restore_auto(self, timeout: float | None = None) -> CmdResult | None: + """把**全部**风扇位交回 BMC 自动控制。 + + 这是进程退出路径上的安全回退动作,必须在任何异常/信号处理里都能跑通, + 因此这里**吞掉所有异常**(只记日志)—— 它要是把退出流程本身搞崩, + 风扇就真回不去了。 + """ + logger.warning("开始回退:全部风扇位交回 BMC 自动控制") + try: + self._target = {slot: AUTO for slot in FAN_SLOT_INDEX} + result = self.apply(timeout=timeout) + if result.ok: + logger.info("已回退 BMC 自动控制") + else: + logger.error("回退命令返回非零: %s", result.summary()) + return result + except Exception: + logger.exception( + "❌ 回退 BMC 自动控制失败 —— 请立即手动确认风扇状态!" + ) + return None diff --git a/app/main.py b/app/main.py new file mode 100644 index 0000000..ede3c9b --- /dev/null +++ b/app/main.py @@ -0,0 +1,198 @@ +"""FastAPI 应用入口。 + +单进程承担三件事: + +1. **控制回路** —— 后台 asyncio 任务(``controller.run()``) +2. **API 服务** —— REST + WebSocket(``api.router``) +3. **静态托管** —— 直接把前端构建产物挂在根路径 + +进程退出路径是这里最需要小心的地方:``lifespan`` 的 ``finally`` 必须 +无条件执行 ``guard.close()``。uvicorn 自己会接 SIGTERM 并触发优雅关闭, +所以我们不抢它的信号处理,只依靠 ``atexit``(``SafetyGuard.engage`` 时 +注册)作为第二道保险。真正的兜底是那个独立的心跳看门狗 —— +SIGKILL 和断电在进程内是不可能捕获的。 +""" + +from __future__ import annotations + +import asyncio +import logging +from contextlib import asynccontextmanager, suppress +from pathlib import Path + +from fastapi import FastAPI +from fastapi.staticfiles import StaticFiles + +from . import __version__ +from .api import router as api_router +from .config import load_config +from .controller import FanController +from .curve import build_curve_from_config +from .ipmi import IPMIClient +from .safety import SafetyGuard +from .sensors import ( + BoardTemperatureReader, + CPUCoreTemperatureReader, + FanMetricsReader, + GPUMetricsReader, +) +from .store import Store + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s %(levelname)-7s [%(name)s] %(message)s", +) +logger = logging.getLogger(__name__) + +#: 前端构建产物目录(``vite build`` 的默认输出) +STATIC_DIR = Path(__file__).resolve().parent / "static" + + +@asynccontextmanager +async def lifespan(app: FastAPI): + config = load_config() + store = Store() + + ipmi = IPMIClient( + binary=config.sources.ipmitool_binary, + remote=config.sources.ipmi_remote, + timeout=config.sources.ipmi_timeout, + ) + guard = SafetyGuard( + ipmi, + heartbeat_path=Path(config.safety.heartbeat_path), + restore_timeout=config.safety.restore_timeout, + ) + gpu_reader = GPUMetricsReader( + dcgm_endpoint=config.sources.dcgm_endpoint, + timeout=config.sources.http_timeout, + ) + fan_reader = FanMetricsReader( + exporter_endpoint=config.sources.ipmi_exporter_endpoint, + ipmi_binary=config.sources.ipmitool_binary, + timeout=config.sources.http_timeout, + ) + curve = build_curve_from_config( + config.curve.points, + hysteresis=config.curve.hysteresis, + min_duty=config.curve.min_duty, + max_duty=config.curve.max_duty, + ) + board_reader = BoardTemperatureReader( + exporter_endpoint=config.sources.ipmi_exporter_endpoint, + ipmi_binary=config.sources.ipmitool_binary, + timeout=config.sources.http_timeout, + ) + cpu_reader = CPUCoreTemperatureReader( + exporter_endpoint=config.sources.node_exporter_endpoint, + ipmi_binary=config.sources.ipmitool_binary, + timeout=config.sources.http_timeout, + ) + controller = FanController( + config, + ipmi, + guard, + gpu_reader, + fan_reader, + curve, + board_reader=board_reader, + cpu_reader=cpu_reader, + store=store, + ) + + # 运行时设置:数据库里的值优先于配置文件(同样是单一数据源)。 + # ⚠️ **必须先于分配恢复** —— 管控范围(managed_gpus)是分配校验的前提, + # 顺序反了的话,未管控卡的存量分配会把整份恢复作废(16:56 事故根因之一)。 + stored_settings = store.load_settings() + if stored_settings: + try: + controller.apply_settings(stored_settings) + logger.info( + "已应用数据库里的运行时设置: %s", ", ".join(sorted(stored_settings)) + ) + except (ValueError, TypeError) as exc: + logger.error("数据库里的设置无效(%s)—— 沿用配置文件里的值", exc) + + # 「散热源 → 风扇位」的分配以数据库为**唯一权威**(配置文件只声明管控哪些位)。 + # + # ⚠️ 库里空的时候**不写任何默认值** —— 首次使用必须由用户在界面上亲自给 + # 每张 GPU 挑风扇接口,程序不替人猜(猜错 = 某张卡散热不足)。在用户完成 + # 之前 controller 保持未配置状态、不调档,前端会弹分配初始化向导。 + stored = store.load_assignments() + if stored: + # 与管控范围冲突的存量分配先自愈(清空冲突项),再整体恢复 —— + # 保证「设置 ↔ 分配」在后端也是一条完整链路 + stored = controller.sanitize_stored_assignments(stored) + try: + controller.update_assignments(stored) + logger.info("已应用数据库里的分配(%d 条)", len(stored)) + except ValueError as exc: + logger.error( + "数据库里的分配无效(%s)—— 保持未配置状态,请在界面上重新分配", exc + ) + else: + logger.warning( + "数据库中还没有分配配置 —— 控制器暂不调档," + "请在界面上完成「风扇分配初始化」(首次使用必做)" + ) + + app.state.config = config + app.state.ipmi = ipmi + app.state.guard = guard + app.state.controller = controller + app.state.store = store + + logger.info( + "GPU 风扇控制台 v%s | 模式=%s | 周期=%.1fs | 预置风扇位: %s(其余靠运行时探测)", + __version__, + config.control.mode, + config.control.interval, + ", ".join(b.slot for b in config.fans) or "(无)", + ) + + # 记一笔启动 —— 「风扇什么时候被谁改过」以后查这张表 + store.log( + "lifecycle", + "startup", + f"服务启动 | 模式={config.control.mode} | 周期={config.control.interval}s" + f" | 管控位={','.join(b.slot for b in config.fans)}" + f" | control.enabled={config.control.enabled}", + ) + + task = asyncio.create_task(controller.run(), name="fan-control-loop") + try: + yield + finally: + logger.info("开始关闭流程") + task.cancel() + with suppress(asyncio.CancelledError): + await task + # 这一步是整个应用最不能跳过的代码 + guard.close("server-shutdown") + guard.clear_heartbeat() + store.log("lifecycle", "shutdown", "服务正常关闭,已把风扇交回 BMC 自动控制") + logger.info("已安全退出") + + +app = FastAPI( + title="GPU Fan Console", + description="pve02 GPU 温度联动风扇控制台(单机应用)", + version=__version__, + lifespan=lifespan, +) + +app.include_router(api_router, prefix="/api") + + +if STATIC_DIR.is_dir(): + app.mount("/", StaticFiles(directory=str(STATIC_DIR), html=True), name="static") +else: + + @app.get("/", include_in_schema=False) + async def _frontend_missing() -> dict[str, str]: + return { + "message": "前端尚未构建", + "hint": "进入 frontend/ 执行 npm install && npm run build," + "产物会输出到 app/static/", + "api_docs": "/docs", + } diff --git a/app/requirements.txt b/app/requirements.txt new file mode 100644 index 0000000..631d420 --- /dev/null +++ b/app/requirements.txt @@ -0,0 +1,4 @@ +fastapi>=0.115.0 +uvicorn[standard]>=0.30.0 +pydantic>=2.7.0 +PyYAML>=6.0 diff --git a/app/safety.py b/app/safety.py new file mode 100644 index 0000000..b49c7fb --- /dev/null +++ b/app/safety.py @@ -0,0 +1,186 @@ +"""安全护栏 —— 本应用最不该省的一块。 + +手动占空比是一条「单行道」:程序异常退出而没回退的话,风扇会**永远停在** +最后一次写入的转速上。BMC 的热保护最终会介入,但中间那段高温窗口足以 +损伤硬件。所以这里的职责只有一件事 —— **保证任何可捕获的退出路径都能把 +风扇交回 BMC 自动控制**。 + +覆盖的退出路径: + +=============== ========================================================== +路径 手段 +=============== ========================================================== +正常结束 ``close()`` / ``with`` 退出 +未捕获异常 ``try/finally`` + ``atexit`` +SIGTERM/SIGINT 信号处理器(systemd stop、Ctrl-C) +SIGKILL / 断电 **捕获不到**,只能靠独立看门狗检查心跳文件兜底 +=============== ========================================================== + +最后一行是设计上的硬伤,不可能在进程内解决 —— 所以心跳文件 + 外部看门狗 +是**必需项**,不是可选优化。配套的看门狗见 ``app/deploy/fan-watchdog.sh``。 +""" + +from __future__ import annotations + +import atexit +import logging +import os +import signal +import threading +import time +from pathlib import Path +from types import FrameType +from typing import Iterable, Sequence + +from .ipmi import GPU_COOLING_SLOTS, IPMIClient + +logger = logging.getLogger(__name__) + +#: 默认心跳超时(秒)。看门狗脚本用同一个值,改这里要顺手改脚本。 +DEFAULT_HEARTBEAT_TIMEOUT: float = 120.0 + + +class SafetyGuard: + """风扇接管状态跟踪 + 回退保证。 + + 用法:: + + guard = SafetyGuard(ipmi, heartbeat_path=Path("/run/fan-console/heartbeat")) + guard.install_signal_handlers() + with guard: + guard.engage() + ipmi.apply({"FRNT_FAN1": 60}) + ... + # 退出 with 时自动回退 + """ + + def __init__( + self, + ipmi: IPMIClient, + heartbeat_path: Path | None = None, + restore_timeout: float = 15.0, + ) -> None: + self._ipmi = ipmi + self._heartbeat_path = heartbeat_path + self._restore_timeout = restore_timeout + + self._lock = threading.RLock() + self._engaged = False + self._closed = False + + # ------------------------------------------------------------ 上下文管理 + + def __enter__(self) -> "SafetyGuard": + return self + + def __exit__(self, exc_type, exc, tb) -> bool: + reason = "normal-exit" if exc_type is None else f"exception-{exc_type.__name__}" + self.close(reason=reason) + return False # 不吞异常,让它继续往上冒 + + # ------------------------------------------------------------ 接管与回退 + + @property + def engaged(self) -> bool: + """是否已经接管过风扇(即写过手动占空比)。""" + return self._engaged + + @property + def closed(self) -> bool: + return self._closed + + def engage(self) -> None: + """标记「开始接管风扇」,并注册 ``atexit`` 兜底。 + + 必须在**第一次写入手动占空比之前**调用,否则进程崩溃时护栏不会触发。 + """ + with self._lock: + if self._engaged: # 幂等,别重复注册 atexit + return + self._engaged = True + atexit.register(self.close, "atexit") + logger.info("安全护栏已激活:进程退出时将把风扇交回 BMC 自动控制") + + def close(self, reason: str = "normal-exit") -> None: + """回退到 BMC 自动控制。**幂等**,可以随便重复调用。""" + with self._lock: + if self._closed: + return + self._closed = True + was_engaged = self._engaged + + if not was_engaged: + logger.debug("未接管过风扇,无需回退(原因: %s)", reason) + return + + logger.warning("触发安全回退(原因: %s)", reason) + self._ipmi.restore_auto(timeout=self._restore_timeout) + + # ------------------------------------------------------------ 信号处理 + + def install_signal_handlers( + self, signals: Sequence[signal.Signals] = (signal.SIGTERM, signal.SIGINT) + ) -> None: + """接管 SIGTERM / SIGINT,先回退再退出。 + + 注意 ``SIGKILL`` 无法捕获 —— 那正是需要外部看门狗的原因。 + """ + for sig in signals: + signal.signal(sig, self._make_handler(sig)) + logger.debug("已接管信号: %s", ", ".join(s.name for s in signals)) + + def _make_handler(self, sig: signal.Signals): + def handler(signum: int, frame: FrameType | None) -> None: + name = signal.Signals(signum).name + logger.warning("收到 %s,执行安全回退", name) + self.close(reason=f"signal-{name}") + # 复位为默认处理再把信号重发给自己,让进程以标准语义终止 + # (保留正确的退出码,systemd 那边才判断得准) + signal.signal(signum, signal.SIG_DFL) + os.kill(os.getpid(), signum) + + return handler + + # ------------------------------------------------------------ 心跳 + + def beat(self, note: str = "") -> None: + """更新心跳文件。 + + 独立看门狗据此判断本进程是否还活着。写的是「原子替换」的文件, + 避免看门狗读到写了一半的内容。 + """ + if self._heartbeat_path is None: + return + try: + self._heartbeat_path.parent.mkdir(parents=True, exist_ok=True) + tmp = self._heartbeat_path.with_name(self._heartbeat_path.name + ".tmp") + tmp.write_text(f"{time.time():.3f}\n{note}\n", encoding="utf-8") + tmp.replace(self._heartbeat_path) + except OSError: + # 心跳写不出去不该拖垮控制回路,但要留痕 + logger.exception("写心跳文件失败: %s", self._heartbeat_path) + + def clear_heartbeat(self) -> None: + """进程正常退出时清掉心跳,看门狗见不到文件就不会误触发。""" + if self._heartbeat_path is None: + return + try: + self._heartbeat_path.unlink(missing_ok=True) + except OSError: + logger.debug("清理心跳文件失败: %s", self._heartbeat_path) + + # ------------------------------------------------------------ 紧急动作 + + def force_full_speed(self, slots: Iterable[str] = GPU_COOLING_SLOTS) -> bool: + """紧急全速。温度失控或传感器读不到时的最后手段。 + + 这是**加**散热方向的操作,不存在「调错会烧硬件」的风险, + 所以允许在检测到异常时自动触发。 + """ + try: + logger.error("🚨 触发紧急全速: %s", ", ".join(slots)) + result = self._ipmi.apply({slot: 100 for slot in slots}) + return bool(result.ok) + except Exception: + logger.exception("紧急全速执行失败") + return False diff --git a/app/sensors.py b/app/sensors.py new file mode 100644 index 0000000..32c2681 --- /dev/null +++ b/app/sensors.py @@ -0,0 +1,632 @@ +"""传感器读取 —— GPU 指标、风扇转速、板载温度。 + +设计取舍: + +- **GPU 温度优先走本地 DCGM 端点**(``127.0.0.1:9400``),而不是解析 + ``nvidia-smi`` 的文本输出。理由:口径与观测侧(Prometheus)完全一致, + 不会出现「控制器用一个值、看板上是另一个值」这种对不上的情况; + 且 Prometheus 文本格式比 nvidia-smi 的人类可读输出稳定得多。 + ``nvidia-smi`` 保留为兜底,DCGM 挂了也不至于瞎眼。 +- **风扇转速优先走本地 ipmi_exporter**(``127.0.0.1:9290``),兜底 ``ipmitool``。 +- **只用标准库**(``urllib``),不引入 ``requests``/``prometheus_client`` —— + 这台机器上少一个依赖就少一个将来炸的点。 +- 所有网络/子进程调用**带超时**。上游项目的教训:一个不带超时的阻塞读 + 能把整个控制线程静默挂死。 + +关于 pve02 这块板子(ASRock Rack EPYCD8,BMC 固件 2.20)的实测要点: + +- ``ipmitool sdr type fan`` 输出**五列**:``名称 | 传感器ID | 状态 | 阈值 | 读数``。 + 读数在**最后一列**。最初按「第二列是读数」写会把传感器 ID ``62h`` 当成 RPM。 +- 风扇传感器有 **14 个位**,其中 10 个是 ``FRNT_FAN2_2`` 这类未接位,全报 + ``No Reading``。它们不参与控制,解析时直接跳过。 +- **BMC 里没有任何 GPU 温度传感器**(实测只有 MB / Card Side / CPU / TR1 / + DDR4_A~H)。这不是漏配,是硬件层面就没接 —— 所以 GPU 联动只能走 DCGM。 +""" + +from __future__ import annotations + +import logging +import re +import subprocess +import urllib.error +import urllib.request +from dataclasses import dataclass +from typing import Any + +logger = logging.getLogger(__name__) + +# ------------------------------------------------------------------ 常量 + +#: ipmi_exporter 的传感器状态码语义(风扇与温度共用同一套编码) +SENSOR_STATE_LABELS: dict[int, str] = { + 0: "nominal", + 1: "warning", + 2: "critical", +} + +# ------------------------------------------------------------------ 解析工具 + +_SAMPLE_RE = re.compile( + r"^(?P[a-zA-Z_:][a-zA-Z0-9_:]*)" + r"(?:\{(?P[^}]*)\})?" + r"[ \t]+(?P[^\s]+)" +) +_LABEL_RE = re.compile(r'([a-zA-Z_][a-zA-Z0-9_]*)="((?:[^"\\]|\\.)*)"') +_RPM_RE = re.compile(r"(\d+)\s*RPM") +_TEMP_RE = re.compile(r"(-?\d+)\s*degrees\s*C", re.IGNORECASE) + + +@dataclass(frozen=True) +class Sample: + """一条 Prometheus 样本。""" + + name: str + labels: dict[str, str] + value: float + + +def parse_prometheus(text: str) -> list[Sample]: + """解析 Prometheus 文本格式。忽略注释与无法解析的行。""" + samples: list[Sample] = [] + for raw_line in text.splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + match = _SAMPLE_RE.match(line) + if not match: + continue + try: + value = float(match.group("value")) + except ValueError: + continue # NaN / +Inf 之类,本项目用不上 + labels = dict(_LABEL_RE.findall(match.group("labels") or "")) + samples.append(Sample(match.group("name"), labels, value)) + return samples + + +def _fetch_text(url: str, timeout: float) -> str: + request = urllib.request.Request( + url, headers={"User-Agent": "gpu-fan-console/0.1"} + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + return response.read().decode("utf-8", errors="replace") + + +def parse_sdr_line(line: str) -> tuple[str, str, str] | None: + """解析一行 ``ipmitool sdr`` 输出,返回 ``(名称, 读数, 状态)``。 + + EPYCD8 上的实际格式是**五列**:``名称 | 传感器ID | 状态 | 阈值 | 读数``。 + 注意**读数在最后一列**、不是第二列(第二列是传感器 ID):: + + FRNT_FAN1 | 62h | ok | 7.0 | 3000 RPM + MB Temp | 31h | ok | 3.0 | 34 degrees C + FRNT_FAN2 | 63h | ns | 7.0 | No Reading + + 2026-09-28 实测踩坑:最初按「第二列是读数」写正则,结果把 ``62h`` + 当成了 RPM,静默解析出 None。不同 ipmitool 版本与传感器类型的列数 + 并不一致,所以按管道分割后取「首列=名称、第三列=状态、末列=读数」, + 比写死正则稳。 + + Returns: + ``(name, reading, state)``;无法解析时返回 ``None``。 + """ + parts = [p.strip() for p in line.split("|")] + if len(parts) < 4 or not parts[0]: + return None + state = parts[2] if len(parts) > 2 else "" + return parts[0], parts[-1], state + + +# ------------------------------------------------------------------ 数据模型 + + +@dataclass +class GPUMetric: + """单张 GPU 的一次快照。""" + + uuid: str + index: int + pci_bus_id: str = "" + model_name: str = "" + temperature: float | None = None + power_watts: float | None = None + utilization: float | None = None + memory_used_mib: float | None = None + memory_total_mib: float | None = None + source: str = "" + + @property + def short_uuid(self) -> str: + """``GPU-e49ed30f-...`` → ``e49ed30f``,用于日志和前端展示。""" + return self.uuid.removeprefix("GPU-")[:8] + + @property + def memory_percent(self) -> float | None: + if not self.memory_total_mib: + return None + used = self.memory_used_mib or 0.0 + return round(used / self.memory_total_mib * 100, 1) + + +@dataclass +class FanReading: + """单个风扇位的一次读数。""" + + slot: str + rpm: float | None = None + state: str = "" + source: str = "" + + +@dataclass +class BoardTemperature: + """BMC 板载温度读数。""" + + name: str + celsius: float | None = None + state: str = "" + + +# ------------------------------------------------------------------ GPU 读取 + + +class GPUMetricsReader: + """GPU 指标读取器:DCGM 端点优先,nvidia-smi 兜底。""" + + def __init__( + self, + dcgm_endpoint: str | None = "http://127.0.0.1:9400/metrics", + nvidia_smi: str = "nvidia-smi", + timeout: float = 5.0, + ) -> None: + self.dcgm_endpoint = dcgm_endpoint + self.nvidia_smi = nvidia_smi + self.timeout = timeout + #: 上一次成功读取用的数据源,便于排障 + self.last_source: str = "" + + # ---------------------------------------------------------- 入口 + + def read(self) -> list[GPUMetric]: + """读取全部 GPU 快照。两个数据源都失败时返回空列表(不抛异常)。""" + if self.dcgm_endpoint: + try: + metrics = self._read_dcgm() + if metrics: + self.last_source = "dcgm" + return metrics + logger.warning("DCGM 端点返回空指标,回退 nvidia-smi") + except (urllib.error.URLError, OSError, ValueError) as exc: + logger.warning("DCGM 端点读取失败,回退 nvidia-smi: %s", exc) + + try: + metrics = self._read_nvidia_smi() + self.last_source = "nvidia-smi" + return metrics + except (OSError, ValueError, subprocess.SubprocessError) as exc: + logger.error("nvidia-smi 也读不到: %s", exc) + return [] + + # ---------------------------------------------------------- DCGM + + #: DCGM 字段名 → :class:`GPUMetric` 属性名 + _DCGM_FIELDS: dict[str, str] = { + "DCGM_FI_DEV_GPU_TEMP": "temperature", + "DCGM_FI_DEV_POWER_USAGE": "power_watts", + "DCGM_FI_DEV_GPU_UTIL": "utilization", + "DCGM_FI_DEV_FB_USED": "memory_used_mib", + # FB_TOTAL 在部分版本里不直接提供,用 USED + FREE 兜底 + "DCGM_FI_DEV_FB_FREE": "_memory_free_mib", + } + + def _read_dcgm(self) -> list[GPUMetric]: + text = _fetch_text(self.dcgm_endpoint or "", self.timeout) + samples = parse_prometheus(text) + + buckets: dict[str, dict[str, Any]] = {} + for sample in samples: + attr = self._DCGM_FIELDS.get(sample.name) + if attr is None or sample.name.startswith("DCGM_FI_PROF_"): + continue + uuid = sample.labels.get("UUID") + if not uuid: + continue + bucket = buckets.setdefault( + uuid, + { + "uuid": uuid, + "index": _safe_int(sample.labels.get("gpu"), -1), + "pci_bus_id": sample.labels.get("pci_bus_id", ""), + "model_name": sample.labels.get("modelName", ""), + }, + ) + bucket[attr] = sample.value + + metrics: list[GPUMetric] = [] + for bucket in buckets.values(): + free = bucket.pop("_memory_free_mib", None) + used = bucket.get("memory_used_mib") + if used is not None and free is not None: + bucket["memory_total_mib"] = used + free + metrics.append(GPUMetric(source="dcgm", **bucket)) + + metrics.sort(key=lambda m: m.index) + return metrics + + # ---------------------------------------------------------- nvidia-smi + + _SMI_FIELDS = ( + "index,uuid,pci.bus_id,name," + "temperature.gpu,power.draw,utilization.gpu,memory.used,memory.total" + ) + + def _read_nvidia_smi(self) -> list[GPUMetric]: + cmd = [ + self.nvidia_smi, + f"--query-gpu={self._SMI_FIELDS}", + "--format=csv,noheader,nounits", + ] + proc = subprocess.run( + cmd, capture_output=True, text=True, timeout=self.timeout, check=False + ) + if proc.returncode != 0: + raise ValueError(f"nvidia-smi 返回 {proc.returncode}: {proc.stderr.strip()}") + + metrics: list[GPUMetric] = [] + for line in proc.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) < 9: + continue + metrics.append( + GPUMetric( + uuid=parts[1], + index=_safe_int(parts[0], -1), + pci_bus_id=parts[2], + model_name=parts[3], + temperature=_safe_float(parts[4]), + power_watts=_safe_float(parts[5]), + utilization=_safe_float(parts[6]), + memory_used_mib=_safe_float(parts[7]), + memory_total_mib=_safe_float(parts[8]), + source="nvidia-smi", + ) + ) + return metrics + + +# ------------------------------------------------------------------ 风扇读取 + + +class FanMetricsReader: + """风扇转速读取器:ipmi_exporter 优先,ipmitool 兜底。""" + + def __init__( + self, + exporter_endpoint: str | None = "http://127.0.0.1:9290/metrics", + ipmi_binary: str = "ipmitool", + timeout: float = 5.0, + ) -> None: + self.exporter_endpoint = exporter_endpoint + self.ipmi_binary = ipmi_binary + self.timeout = timeout + self.last_source: str = "" + + def read(self) -> dict[str, FanReading]: + """读取全部风扇位转速,返回 ``{风扇位名: FanReading}``。""" + if self.exporter_endpoint: + try: + readings = self._read_exporter() + if readings: + self.last_source = "ipmi_exporter" + return readings + except (urllib.error.URLError, OSError, ValueError) as exc: + logger.warning("ipmi_exporter 读取失败,回退 ipmitool: %s", exc) + + try: + readings = self._read_ipmitool() + self.last_source = "ipmitool" + return readings + except (OSError, ValueError, subprocess.SubprocessError) as exc: + logger.error("ipmitool 也读不到风扇转速: %s", exc) + return {} + + def _read_exporter(self) -> dict[str, FanReading]: + text = _fetch_text(self.exporter_endpoint or "", self.timeout) + rpms: dict[str, float] = {} + states: dict[str, int] = {} + + for sample in parse_prometheus(text): + name = sample.labels.get("name") + if not name: + continue + if sample.name == "ipmi_fan_speed_rpm": + rpms[name] = sample.value + elif sample.name == "ipmi_fan_speed_state": + states[name] = int(sample.value) + + readings: dict[str, FanReading] = {} + for name, rpm in rpms.items(): + code = states.get(name, 0) + readings[name] = FanReading( + slot=name, + rpm=rpm, + state=SENSOR_STATE_LABELS.get(code, f"code={code}"), + source="ipmi_exporter", + ) + return readings + + def _read_ipmitool(self) -> dict[str, FanReading]: + proc = subprocess.run( + [self.ipmi_binary, "sdr", "type", "fan"], + capture_output=True, + text=True, + timeout=self.timeout, + check=False, + ) + if proc.returncode != 0: + raise ValueError(f"ipmitool 返回 {proc.returncode}") + + readings: dict[str, FanReading] = {} + for line in proc.stdout.splitlines(): + parsed = parse_sdr_line(line.strip()) + if parsed is None: + continue + slot, reading, state = parsed + rpm_match = _RPM_RE.search(reading) + if rpm_match is None: + # EPYCD8 会报出一堆**未接**的传感器位(``FRNT_FAN2_2`` 之类, + # 实测共 14 个位、其中 10 个是 No Reading)。它们不参与控制, + # 跳过 —— 也让两条数据源的行为保持一致(ipmi_exporter 只暴露 + # 有读数的 4 个位)。 + continue + readings[slot] = FanReading( + slot=slot, + rpm=float(rpm_match.group(1)), + state=state, + source="ipmitool", + ) + return readings + + +# ------------------------------------------------------------------ 板载温度 + + +class BoardTemperatureReader: + """BMC 板载温度读取(MB / CPU / Card Side / DDR4_*)。 + + ⚠️ **这些全都不是 GPU 温度。** EPYCD8 的 BMC 里没有任何 GPU 温度传感器 —— + 这不是漏配,是硬件层面就没接。所以 GPU 联动必须走 DCGM,BMC 那条 + 11 级自动温度-占空比曲线对 GPU 完全无效。 + + 这里读板载温度纯粹是给界面多一个参照,尤其 ``Card Side Temp`` + (相对最能反映机箱内扩展卡区域的热环境)。 + """ + + def __init__( + self, + exporter_endpoint: str | None = "http://127.0.0.1:9290/metrics", + ipmi_binary: str = "ipmitool", + timeout: float = 5.0, + ) -> None: + self.exporter_endpoint = exporter_endpoint + self.ipmi_binary = ipmi_binary + self.timeout = timeout + self.last_source: str = "" + + def read(self) -> dict[str, BoardTemperature]: + if self.exporter_endpoint: + try: + readings = self._read_exporter() + if readings: + self.last_source = "ipmi_exporter" + return readings + except (urllib.error.URLError, OSError, ValueError) as exc: + logger.warning("板载温度读取失败(exporter),回退 ipmitool: %s", exc) + + try: + readings = self._read_ipmitool() + self.last_source = "ipmitool" + return readings + except (OSError, ValueError, subprocess.SubprocessError) as exc: + logger.error("板载温度读取失败(ipmitool): %s", exc) + return {} + + def _read_exporter(self) -> dict[str, BoardTemperature]: + text = _fetch_text(self.exporter_endpoint or "", self.timeout) + values: dict[str, float] = {} + states: dict[str, int] = {} + + for sample in parse_prometheus(text): + name = sample.labels.get("name") + if not name: + continue + if sample.name == "ipmi_temperature_celsius": + values[name] = sample.value + elif sample.name == "ipmi_temperature_state": + states[name] = int(sample.value) + + return { + name: BoardTemperature( + name=name, + celsius=celsius, + state=SENSOR_STATE_LABELS.get(states.get(name, 0), "unknown"), + ) + for name, celsius in values.items() + } + + def _read_ipmitool(self) -> dict[str, BoardTemperature]: + proc = subprocess.run( + [self.ipmi_binary, "sdr", "type", "Temperature"], + capture_output=True, + text=True, + timeout=self.timeout, + check=False, + ) + if proc.returncode != 0: + raise ValueError(f"ipmitool 返回 {proc.returncode}") + + readings: dict[str, BoardTemperature] = {} + for line in proc.stdout.splitlines(): + parsed = parse_sdr_line(line.strip()) + if parsed is None: + continue + name, reading, state = parsed + temp_match = _TEMP_RE.search(reading) + if temp_match is None: + continue # "No Reading"(未接的 DDR4 槽位等) + readings[name] = BoardTemperature( + name=name, celsius=float(temp_match.group(1)), state=state + ) + return readings + + +# ------------------------------------------------------------------ CPU 核温度 + + +@dataclass +class CPUCoreTemperature: + """CPU 核心温度(node_exporter 的 hwmon collector)。""" + + label: str + celsius: float + chip: str = "" + + +class CPUCoreTemperatureReader: + """CPU 核温度读取(node_exporter 的 hwmon collector)。 + + ⚠️ **k10temp 在 Prometheus/hwmon 里的 chip 名是 PCI 路径形式** + (``pci0000:00_0000:00:18_3``),不是可读的 ``k10temp`` —— 因为 AMD 的 + k10temp 挂在 PCI 设备 ``00:18.3`` 下。硬编码 chip 名换台机器就废了, + 所以这里靠 ``node_hwmon_sensor_label`` 做**语义关联**: + + - ``node_hwmon_temp_celsius{chip, sensor}`` → 数值 + - ``node_hwmon_sensor_label{chip, sensor, label}`` → 可读名(Tctl / Tccd1…) + + 两个指标都在 node_exporter 的 ``/metrics`` 里,直接从本机端点解析, + 不绕 Prometheus(那边是 30s 快照)。 + """ + + def __init__( + self, + exporter_endpoint: str | None = "http://127.0.0.1:9100/metrics", + ipmi_binary: str = "ipmitool", + timeout: float = 5.0, + ) -> None: + self.exporter_endpoint = exporter_endpoint + self.ipmi_binary = ipmi_binary + self.timeout = timeout + self.last_source: str = "" + + def read(self) -> list[CPUCoreTemperature]: + if self.exporter_endpoint: + try: + readings = self._read_exporter() + if readings: + self.last_source = "node_exporter" + return readings + except (urllib.error.URLError, OSError, ValueError) as exc: + logger.warning("CPU 核温度读取失败(node_exporter): %s", exc) + + try: + readings = self._read_sensors_command() + self.last_source = "sensors" + return readings + except (OSError, subprocess.SubprocessError) as exc: + logger.warning("CPU 核温度读取失败(sensors): %s", exc) + return [] + + def _read_exporter(self) -> list[CPUCoreTemperature]: + text = _fetch_text(self.exporter_endpoint or "", self.timeout) + samples = parse_prometheus(text) + + values: dict[tuple[str, str], float] = {} + labels: dict[tuple[str, str], str] = {} + + for sample in samples: + chip = sample.labels.get("chip", "") + sensor = sample.labels.get("sensor", "") + if not chip or not sensor: + continue + if sample.name.startswith("node_hwmon_temp_celsius"): + values[(chip, sensor)] = sample.value + elif sample.name.startswith("node_hwmon_sensor_label"): + labels[(chip, sensor)] = sample.labels.get("label", "") + + readings: list[CPUCoreTemperature] = [] + for (chip, sensor), celsius in values.items(): + label = labels.get((chip, sensor), "") + # 只保留 CPU 核心相关的(Tctl / Tccd*),别把 NVMe、网卡、 + # 主板 SuperIO 的温度也混进来 —— 那些由板载温度那一路负责 + if not (label.startswith("Tctl") or label.startswith("Tccd")): + continue + readings.append( + CPUCoreTemperature(label=label, celsius=celsius, chip=chip) + ) + + # Tctl 排最前(它才是控速真正关心的),其余按名称排 + readings.sort(key=lambda r: (r.label != "Tctl", r.label)) + return readings + + def _read_sensors_command(self) -> list[CPUCoreTemperature]: + """兜底:解析 ``sensors`` 命令输出。 + + ⚠️ 只当 node_exporter 不可用时用 —— ``sensors`` 的输出是给人看的 + 排版,还混着大量无效项(未接的传感器脚会报 ``ALARM``),不适合 + 程序解析。这里只挑 k10temp 那一段的 Tctl/Tccd。 + """ + proc = subprocess.run( + ["sensors"], + capture_output=True, + text=True, + timeout=self.timeout, + check=False, + ) + if proc.returncode != 0: + raise ValueError(f"sensors 返回 {proc.returncode}") + + readings: list[CPUCoreTemperature] = [] + in_k10temp = False + for line in proc.stdout.splitlines(): + if line.startswith("k10temp-"): + in_k10temp = True + continue + if not in_k10temp: + continue + if line and not line.startswith((" ", "\t")): + break # 离开 k10temp 段落 + stripped = line.strip() + if not stripped: + continue + head, _, rest = stripped.partition(":") + label = head.strip() + if not (label.startswith("Tctl") or label.startswith("Tccd")): + continue + match = re.search(r"([+-]?\d+(?:\.\d+)?)", rest) + if match: + readings.append( + CPUCoreTemperature(label=label, celsius=float(match.group(1))) + ) + + readings.sort(key=lambda r: (r.label != "Tctl", r.label)) + return readings + + +# ------------------------------------------------------------------ 小工具 + + +def _safe_float(value: str | None) -> float | None: + if value is None: + return None + text = value.strip() + if not text or text.lower() in {"n/a", "na", "[n/a]", "nan", "not supported"}: + return None + try: + return float(text) + except ValueError: + return None + + +def _safe_int(value: str | None, default: int = 0) -> int: + parsed = _safe_float(value) + return default if parsed is None else int(parsed) diff --git a/app/store.py b/app/store.py new file mode 100644 index 0000000..0c8dfef --- /dev/null +++ b/app/store.py @@ -0,0 +1,255 @@ +"""SQLite 持久化 —— 绑定关系 + 操作审计。 + +**为什么用 SQLite**:``sqlite3`` 是 Python 标准库,**零额外依赖**(这个项目到 +现在只有 PyYAML 一个第三方依赖,想守住这条线);同时它比 JSON 文件规范得多 —— +有事务、有类型、能查询,将来要扩展(曲线历史、按时间检索审计)也不用改结构。 + +存两类东西: + +1. **``fan_bindings``** —— 风扇位 ↔ 温度源的绑定。用户运行时改的绑定得扛得住重启。 +2. **``audit_log``** —— 操作审计。**这张表是今天被逼出来的**:两个 GPU 风扇位在 + 11:34~11:59 之间从 3000 RPM 掉回 BMC 自动档,结果 uvicorn 日志被重启覆盖、 + IPMI raw 命令又不进 BMC SEL,**翻遍两边都没查出是谁发的命令**。有了审计表, + 这类「到底谁改的」问题以后直接查库。 + +并发注意:``sqlite3`` 的连接不能跨线程共享,而本应用有 asyncio 工作线程 +(``asyncio.to_thread``)。所以这里**每次操作开一个新连接**(SQLite 打开极快), +再配 WAL 模式提升读写并发。写入频率本来就很低,不必上连接池。 +""" + +from __future__ import annotations + +import json +import logging +import sqlite3 +import time +from contextlib import contextmanager +from pathlib import Path +from typing import Any, Iterator + +logger = logging.getLogger(__name__) + +#: 默认数据库位置 +DEFAULT_DB_PATH = Path(__file__).resolve().parent / "data" / "fan-console.db" + +_SCHEMA = """ +CREATE TABLE IF NOT EXISTS fan_assignments ( + source_key TEXT PRIMARY KEY, -- "gpu:" / "gpu:all" / "cpu" + kind TEXT NOT NULL, -- gpu / gpu_group / cpu + gpu_uuid TEXT, -- kind=gpu 时的 GPU UUID(绝不用 index) + slots TEXT NOT NULL DEFAULT '[]', -- 分配给该源的风扇位(JSON 数组) + updated_at REAL NOT NULL +); + +CREATE TABLE IF NOT EXISTS audit_log ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + ts REAL NOT NULL, + kind TEXT NOT NULL, + actor TEXT NOT NULL, + detail TEXT NOT NULL DEFAULT '', + ok INTEGER NOT NULL DEFAULT 1 +); + +CREATE INDEX IF NOT EXISTS idx_audit_ts ON audit_log(ts DESC); + +CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL, + updated_at REAL NOT NULL +); +""" + + +class Store: + """极简持久化层。所有方法都是线程安全的(每次开新连接)。""" + + def __init__(self, path: Path | str | None = None) -> None: + self.path = Path(path) if path else DEFAULT_DB_PATH + self.path.parent.mkdir(parents=True, exist_ok=True) + self._init_schema() + + # ------------------------------------------------------------ 基础设施 + + @contextmanager + def _conn(self) -> Iterator[sqlite3.Connection]: + conn = sqlite3.connect(self.path, timeout=5.0) + conn.row_factory = sqlite3.Row + try: + yield conn + conn.commit() + except Exception: + conn.rollback() + raise + finally: + conn.close() + + def _init_schema(self) -> None: + with self._conn() as conn: + # WAL:读写并发更友好,掉电安全性也更好 + conn.execute("PRAGMA journal_mode=WAL") + conn.executescript(_SCHEMA) + logger.info("持久化已就绪: %s", self.path) + + # ------------------------------------------------------------ 分配 + + def load_assignments(self) -> list[dict[str, Any]] | None: + """读取全部「散热源 → 风扇位」分配。 + + 表中无记录时返回 ``None``(首次使用,前端应弹分配向导)。 + """ + try: + with self._conn() as conn: + rows = conn.execute( + "SELECT source_key, kind, gpu_uuid, slots FROM fan_assignments" + ).fetchall() + except sqlite3.Error as exc: + logger.warning("读取分配失败: %s", exc) + return None + + if not rows: + return None + + result: list[dict[str, Any]] = [] + for row in rows: + try: + slots = json.loads(row["slots"]) + except (json.JSONDecodeError, TypeError): + slots = [] + result.append( + { + "key": row["source_key"], + "kind": row["kind"], + "gpu_uuid": row["gpu_uuid"], + "slots": slots, + } + ) + return result + + def save_assignments(self, assignments: list[dict[str, Any]]) -> None: + """整体覆盖分配(一个事务内完成,不会出现改了一半的状态)。""" + now = time.time() + with self._conn() as conn: + # 先清空:分配是「全量语义」,不在列表里的源就等于未分配 + conn.execute("DELETE FROM fan_assignments") + conn.executemany( + "INSERT INTO fan_assignments (source_key, kind, gpu_uuid, slots, updated_at) " + "VALUES (?, ?, ?, ?, ?)", + [ + ( + item.get("key", ""), + item.get("kind", "gpu"), + item.get("gpu_uuid"), + json.dumps(item.get("slots") or [], ensure_ascii=False), + now, + ) + for item in assignments + ], + ) + logger.info("分配已持久化(%d 条)", len(assignments)) + + # ------------------------------------------------------------ 审计 + + def log( + self, + kind: str, + actor: str, + detail: str = "", + ok: bool = True, + ) -> None: + """记一条审计。 + + Args: + kind: 类别 —— ``ipmi_write`` / ``api_call`` / ``lifecycle`` / ``error`` + actor: 触发者 —— ``auto_curve`` / ``manual`` / ``restore_auto`` / + ``emergency`` / ``watchdog`` / ``api`` / ``startup`` / ``shutdown`` + detail: 人话描述(会被记进库,方便回溯) + ok: 是否成功 + + 审计**绝不能反过来影响主流程** —— 所以这里吞掉所有异常,只记日志。 + """ + try: + with self._conn() as conn: + conn.execute( + "INSERT INTO audit_log (ts, kind, actor, detail, ok) VALUES (?, ?, ?, ?, ?)", + (time.time(), kind, actor, detail[:500], 1 if ok else 0), + ) + except sqlite3.Error: + logger.exception("写审计失败(不影响主流程)") + + def recent_audit(self, limit: int = 100) -> list[dict[str, Any]]: + """最近的审计记录(新的在前)。""" + try: + with self._conn() as conn: + rows = conn.execute( + "SELECT id, ts, kind, actor, detail, ok FROM audit_log " + "ORDER BY ts DESC LIMIT ?", + (limit,), + ).fetchall() + except sqlite3.Error as exc: + logger.warning("读取审计失败: %s", exc) + return [] + + return [ + { + "id": row["id"], + "ts": row["ts"], + "kind": row["kind"], + "actor": row["actor"], + "detail": row["detail"], + "ok": bool(row["ok"]), + } + for row in rows + ] + + def prune_audit(self, keep_days: int = 30) -> int: + """清理过期审计,返回删除条数。""" + cutoff = time.time() - keep_days * 86400 + with self._conn() as conn: + cursor = conn.execute("DELETE FROM audit_log WHERE ts < ?", (cutoff,)) + deleted = cursor.rowcount + if deleted: + logger.info("清理了 %d 条过期审计(保留 %d 天)", deleted, keep_days) + return deleted + + # ------------------------------------------------------------ 运行时设置 + + def load_settings(self) -> dict[str, Any]: + """读取全部运行时设置(value 是 JSON,已反序列化)。 + + 返回空 dict 表示「还没有任何运行时设置」—— 调用方应沿用配置文件里的值。 + """ + try: + with self._conn() as conn: + rows = conn.execute("SELECT key, value FROM settings").fetchall() + except sqlite3.Error as exc: + logger.warning("读取设置失败,沿用配置文件: %s", exc) + return {} + + result: dict[str, Any] = {} + for row in rows: + try: + result[row["key"]] = json.loads(row["value"]) + except (json.JSONDecodeError, TypeError): + logger.warning("设置项 %s 的值不是合法 JSON,已忽略", row["key"]) + return result + + def save_settings(self, settings: dict[str, Any]) -> None: + """写入/更新设置。 + + **部分更新语义** —— 只覆盖传进来的键,没传的保持原样。 + (绑定那张表相反,是全量覆盖,因为绑定本身就是一整份配置。) + """ + if not settings: + return + now = time.time() + with self._conn() as conn: + conn.executemany( + "INSERT INTO settings (key, value, updated_at) VALUES (?, ?, ?) " + "ON CONFLICT(key) DO UPDATE SET value=excluded.value, " + "updated_at=excluded.updated_at", + [ + (key, json.dumps(value, ensure_ascii=False), now) + for key, value in settings.items() + ], + ) + logger.info("设置已持久化: %s", ", ".join(sorted(settings))) diff --git a/app/tests/__init__.py b/app/tests/__init__.py new file mode 100644 index 0000000..aa9c0f5 --- /dev/null +++ b/app/tests/__init__.py @@ -0,0 +1 @@ +"""app 的测试包。""" diff --git a/app/tests/test_core.py b/app/tests/test_core.py new file mode 100644 index 0000000..315e2bc --- /dev/null +++ b/app/tests/test_core.py @@ -0,0 +1,624 @@ +"""核心逻辑单元测试(只依赖标准库,可离线跑)。 + +重点覆盖两类「出错就烧硬件」的逻辑: + +1. **8 字节 payload 拼装** —— 少一个字节 BMC 会静默忽略整条命令; + 而「必须全量写」又意味着改一个位时不能把其它位踩成 0x00。 +2. **曲线滞回** —— 没有滞回,温度在阈值附近抖动会让风扇转速反复横跳。 + +跑法(在项目根目录):: + + python -m unittest discover -s app/tests -t . -v +""" + +from __future__ import annotations + +import sys +import unittest +from pathlib import Path +from unittest import mock + +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +from app.curve import CurvePoint, CurveState, FanCurve # noqa: E402 +from app.ipmi import ( # noqa: E402 + FAN_SLOT_INDEX, + PAYLOAD_LEN, + RESERVED_INDEX, + IPMIClient, + _fmt_byte, + encode_duty, +) +from app.sensors import ( # noqa: E402 + FanMetricsReader, + parse_prometheus, + parse_sdr_line, +) + + +class TestDutyEncoding(unittest.TestCase): + """占空比 ↔ 字节值的编码。""" + + def test_auto_maps_to_zero(self) -> None: + self.assertEqual(encode_duty(None), 0x00) + + def test_valid_duty_passes_through(self) -> None: + for duty in (1, 20, 50, 100): + self.assertEqual(encode_duty(duty), duty) + + def test_zero_is_rejected(self) -> None: + # 0 被保留用于表示「自动」,不能当手动值下发 + with self.assertRaises(ValueError): + encode_duty(0) + + def test_over_hundred_is_rejected(self) -> None: + with self.assertRaises(ValueError): + encode_duty(101) + + def test_non_int_is_rejected(self) -> None: + with self.assertRaises(TypeError): + encode_duty(50.5) # type: ignore[arg-type] + with self.assertRaises(TypeError): + encode_duty(True) # type: ignore[arg-type] + + def test_hex_formatting_matches_spec(self) -> None: + """字节的十进制值就是百分比 —— 这是最容易写错的一处。""" + self.assertEqual(_fmt_byte(20), "0x14") + self.assertEqual(_fmt_byte(30), "0x1e") + self.assertEqual(_fmt_byte(50), "0x32") + self.assertEqual(_fmt_byte(100), "0x64") + + +class TestPayloadAssembly(unittest.TestCase): + """8 字节 payload 拼装 —— 本文件里最要紧的一组测试。""" + + def setUp(self) -> None: + self.client = IPMIClient(dry_run=True) + + def test_all_auto_is_eight_zeros(self) -> None: + payload = self.client.build_payload() + self.assertEqual(len(payload), PAYLOAD_LEN) + self.assertEqual(payload, [0] * PAYLOAD_LEN) + + def test_length_is_always_eight(self) -> None: + payload = self.client.build_payload({"FRNT_FAN1": 60, "REAR_FAN2": 45}) + self.assertEqual(len(payload), PAYLOAD_LEN) + + def test_single_slot_maps_to_correct_index(self) -> None: + payload = self.client.build_payload({"FRNT_FAN1": 60}) + self.assertEqual(payload[FAN_SLOT_INDEX["FRNT_FAN1"]], 60) + # b5 之外全是 0 + others = [v for i, v in enumerate(payload) if i != FAN_SLOT_INDEX["FRNT_FAN1"]] + self.assertEqual(others, [0] * (PAYLOAD_LEN - 1)) + + def test_reserved_byte_stays_zero(self) -> None: + """b2 是保留位,无论怎么设都必须保持 0x00。""" + payload = self.client.build_payload({"FRNT_FAN1": 80, "REAR_FAN2": 80}) + self.assertEqual(payload[RESERVED_INDEX], 0x00) + + def test_updating_one_slot_keeps_the_other(self) -> None: + """核心用例:BMC 要求全量写,所以改一个位绝不能踩掉另一个位。""" + self.client.build_payload({"FRNT_FAN1": 60}) + payload = self.client.build_payload({"REAR_FAN2": 45}) + + self.assertEqual(payload[FAN_SLOT_INDEX["FRNT_FAN1"]], 60, "FRNT_FAN1 被踩掉了") + self.assertEqual(payload[FAN_SLOT_INDEX["REAR_FAN2"]], 45) + self.assertEqual(payload[FAN_SLOT_INDEX["CPU1_FAN1"]], 0, "未接管的位应保持自动") + + def test_can_release_single_slot_back_to_auto(self) -> None: + self.client.build_payload({"FRNT_FAN1": 60}) + payload = self.client.build_payload({"FRNT_FAN1": None}) + self.assertEqual(payload[FAN_SLOT_INDEX["FRNT_FAN1"]], 0) + + def test_unknown_slot_raises(self) -> None: + with self.assertRaises(ValueError): + self.client.build_payload({"NOT_A_FAN": 50}) + + def test_restore_auto_zeroes_everything(self) -> None: + self.client.build_payload({"FRNT_FAN1": 90, "REAR_FAN2": 90}) + payload = self.client.build_payload( + {slot: None for slot in FAN_SLOT_INDEX} + ) + self.assertEqual(payload, [0] * PAYLOAD_LEN) + + def test_dry_run_apply_does_not_call_ipmitool(self) -> None: + """dry-run 模式下 apply 不该真的调用 ipmitool。""" + result = self.client.apply({"FRNT_FAN1": 55}) + self.assertTrue(result.ok) + self.assertIn("0x3a", result.args) + self.assertIn("0x37", result.args) # 55 → 0x37 + + +class TestFanCurve(unittest.TestCase): + """温度-占空比曲线与滞回。""" + + def setUp(self) -> None: + self.curve = FanCurve( + [ + CurvePoint(50, 40), + CurvePoint(60, 50), + CurvePoint(70, 65), + CurvePoint(80, 85), + CurvePoint(90, 100), + ], + hysteresis=3.0, + min_duty=30, + max_duty=100, + ) + + def test_stateless_lookup(self) -> None: + """阶梯语义:温度**达到**折点才用那一档,未达到就用低一档(偏保守)。""" + self.assertEqual(self.curve.duty_at(45), 40) # 低于最低折点 → 最低档 + self.assertEqual(self.curve.duty_at(50), 40) + self.assertEqual(self.curve.duty_at(59.9), 40) # 还差一点到 60 + self.assertEqual(self.curve.duty_at(60), 50) + self.assertEqual(self.curve.duty_at(75), 65) # 70 ≤ 75 < 80 → 第 2 档 + self.assertEqual(self.curve.duty_at(80), 85) + self.assertEqual(self.curve.duty_at(95), 100) + + def test_first_step_sets_index(self) -> None: + state = CurveState() + self.assertEqual(self.curve.step(72, state), 65) # 70 ≤ 72 < 80 + self.assertEqual(state.index, 2) + + def test_upshift_is_immediate(self) -> None: + """升温必须立即升档 —— 散热是安全方向,不能有任何延迟。""" + state = CurveState() + self.curve.step(55, state) # 第 0 档 + duty = self.curve.step(80, state) # 冲到 80 → 立即到第 3 档 + self.assertEqual(duty, 85) + self.assertEqual(state.index, 3) + + def test_downshift_requires_leaving_hysteresis_band(self) -> None: + """降温方向:跌出滞回带前不许降档。""" + state = CurveState() + self.curve.step(80, state) # 第 3 档(折点 80°C / 85%) + self.assertEqual(state.index, 3) + + # 78°C 还没跌出 80-3=77 的滞回带 → 保持第 3 档 + self.assertEqual(self.curve.step(78, state), 85) + self.assertEqual(state.index, 3) + + # 76°C 已跌出 → 允许降到第 2 档(70°C / 65%) + self.assertEqual(self.curve.step(76, state), 65) + self.assertEqual(state.index, 2) + + def test_no_flapping_around_threshold(self) -> None: + """阈值附近抖动不应导致转速横跳("直升机效应"回归测试)。""" + state = CurveState() + self.curve.step(70.0, state) # 定在第 2 档 + baseline = state.index + + for temp in (69.9, 70.1, 69.5, 70.4, 69.8, 70.2): + self.curve.step(temp, state) + + self.assertEqual(state.index, baseline, "温度微抖导致档位漂移") + + def test_min_duty_clamp(self) -> None: + curve = FanCurve([CurvePoint(50, 10)], min_duty=30) + self.assertEqual(curve.duty_at(30), 30) + + def test_max_duty_clamp(self) -> None: + curve = FanCurve([CurvePoint(50, 100)], max_duty=80) + self.assertEqual(curve.duty_at(99), 80) + + def test_points_are_sorted(self) -> None: + curve = FanCurve([CurvePoint(90, 100), CurvePoint(50, 40)]) + self.assertEqual([p.temp for p in curve.points], [50, 90]) + + def test_duplicate_temp_rejected(self) -> None: + with self.assertRaises(ValueError): + FanCurve([CurvePoint(50, 40), CurvePoint(50, 60)]) + + def test_empty_points_rejected(self) -> None: + with self.assertRaises(ValueError): + FanCurve([]) + + def test_state_reset(self) -> None: + state = CurveState(index=4) + state.reset() + self.assertIsNone(state.index) + + +class TestPrometheusParsing(unittest.TestCase): + """Prometheus 文本解析 —— 用 pve02 上抓到的真实格式。""" + + SAMPLE = """\ +# HELP DCGM_FI_DEV_GPU_TEMP GPU temperature (in C). +# TYPE DCGM_FI_DEV_GPU_TEMP gauge +DCGM_FI_DEV_GPU_TEMP{gpu="0",UUID="GPU-e49ed30f-f0f4-dc17-0225-2c1235602b39",pci_bus_id="00000000:01:00.0",device="nvidia0",modelName="Tesla T10"} 45 +DCGM_FI_DEV_GPU_TEMP{gpu="1",UUID="GPU-e60e8f23-b150-6718-8508-3cb00e0d9fc6",pci_bus_id="00000000:82:00.0",device="nvidia1",modelName="Tesla T10"} 52 +DCGM_FI_DEV_POWER_USAGE{gpu="0",UUID="GPU-e49ed30f-f0f4-dc17-0225-2c1235602b39"} 43.017 +# HELP ipmi_fan_speed_rpm Fan speed in rotations per minute. +ipmi_fan_speed_rpm{id="24",name="FRNT_FAN1"} 3000 +ipmi_fan_speed_rpm{id="29",name="REAR_FAN2"} 3000 +""" + + def test_parses_all_samples(self) -> None: + samples = parse_prometheus(self.SAMPLE) + self.assertEqual(len(samples), 5) + + def test_ignores_comments_and_help(self) -> None: + samples = parse_prometheus(self.SAMPLE) + self.assertFalse([s for s in samples if s.name.startswith("#")]) + + def test_extracts_labels(self) -> None: + samples = parse_prometheus(self.SAMPLE) + temps = [s for s in samples if s.name == "DCGM_FI_DEV_GPU_TEMP"] + self.assertEqual(len(temps), 2) + self.assertEqual(temps[0].labels["gpu"], "0") + self.assertEqual(temps[0].labels["modelName"], "Tesla T10") + self.assertAlmostEqual(temps[0].value, 45.0) + + def test_uuid_label_present(self) -> None: + """UUID 标签是「按卡绑定」的基础,必须解析出来。""" + samples = parse_prometheus(self.SAMPLE) + uuids = { + s.labels["UUID"] + for s in samples + if s.name == "DCGM_FI_DEV_GPU_TEMP" + } + self.assertEqual( + uuids, + { + "GPU-e49ed30f-f0f4-dc17-0225-2c1235602b39", + "GPU-e60e8f23-b150-6718-8508-3cb00e0d9fc6", + }, + ) + + def test_ignores_unparsable_lines(self) -> None: + self.assertEqual(parse_prometheus("garbage line without value"), []) + self.assertEqual(parse_prometheus(""), []) + + +class TestSdrFanParsing(unittest.TestCase): + """``ipmitool sdr type fan`` 输出解析。 + + 样例取自 2026-09-28 在 pve02 上的真实输出 —— 这组用例的由来就是一个 + 真实 bug:最初以为「第二列是读数」,结果把传感器 ID ``62h`` 当成了 RPM。 + """ + + SAMPLE_OK = "FRNT_FAN1 | 62h | ok | 7.0 | 3000 RPM" + SAMPLE_NS = "FRNT_FAN2 | 63h | ns | 7.0 | No Reading" + SAMPLE_CPU = "CPU1_FAN1 | 60h | ok | 7.0 | 1300 RPM" + + def test_reading_comes_from_last_column(self) -> None: + """核心断言:读数是最后一列,不是第二列的传感器 ID。""" + parsed = parse_sdr_line(self.SAMPLE_OK) + self.assertIsNotNone(parsed) + slot, reading, state = parsed + self.assertEqual(slot, "FRNT_FAN1") + self.assertEqual(reading, "3000 RPM") + self.assertEqual(state, "ok") + + def test_extracts_rpm_value(self) -> None: + import re + + _, reading, _ = parse_sdr_line(self.SAMPLE_CPU) + match = re.search(r"(\d+)\s*RPM", reading) + self.assertIsNotNone(match) + self.assertEqual(match.group(1), "1300") + + def test_no_reading_row_is_parsed_without_rpm(self) -> None: + parsed = parse_sdr_line(self.SAMPLE_NS) + self.assertIsNotNone(parsed) + slot, reading, state = parsed + self.assertEqual(slot, "FRNT_FAN2") + self.assertEqual(reading, "No Reading") + self.assertEqual(state, "ns") + + def test_never_mistakes_sensor_id_for_reading(self) -> None: + """回归断言:``62h`` 这种传感器 ID 绝不能被当成读数。""" + _, reading, _ = parse_sdr_line(self.SAMPLE_OK) + self.assertNotIn("h", reading.lower().replace("reading", "")) + + def test_garbage_returns_none(self) -> None: + self.assertIsNone(parse_sdr_line("")) + self.assertIsNone(parse_sdr_line("no pipes here")) + self.assertIsNone(parse_sdr_line("only | one")) + + +class TestSdrTemperatureParsing(unittest.TestCase): + """``ipmitool sdr type Temperature`` 解析(EPYCD8 真实输出)。 + + 与风扇共用同一套五列结构,所以复用 :func:`parse_sdr_line`。 + """ + + SAMPLE = "Card Side Temp | 32h | ok | 3.0 | 47 degrees C" + SAMPLE_NS = "TR1 Temp | 33h | ns | 3.0 | No Reading" + + def test_parses_temperature_row(self) -> None: + parsed = parse_sdr_line(self.SAMPLE) + self.assertIsNotNone(parsed) + name, reading, state = parsed + self.assertEqual(name, "Card Side Temp") + self.assertEqual(reading, "47 degrees C") + self.assertEqual(state, "ok") + + def test_no_reading_temperature(self) -> None: + name, reading, state = parse_sdr_line(self.SAMPLE_NS) + self.assertEqual(name, "TR1 Temp") + self.assertEqual(reading, "No Reading") + self.assertEqual(state, "ns") + + +class TestFanReaderIpmitoolFallback(unittest.TestCase): + """ipmitool 兜底路径 —— 必须跳过未接的风扇位。 + + 样例是 2026-09-28 在 pve02 上抓的真实输出:14 个风扇传感器位里只有 + 4 个有读数,其余全是 ``No Reading``。不跳过的话界面上会凭空多出 + 10 个空风扇位。 + """ + + SDR_OUTPUT = """\ +CPU1_FAN1 | 60h | ok | 7.0 | 1300 RPM +FRNT_FAN1 | 62h | ok | 7.0 | 3000 RPM +FRNT_FAN2 | 63h | ns | 7.0 | No Reading +FRNT_FAN3 | 64h | ns | 7.0 | No Reading +REAR_FAN1 | 66h | ok | 7.0 | 400 RPM +REAR_FAN2 | 67h | ok | 7.0 | 3000 RPM +CPU1_FAN1_2 | 68h | ns | 7.0 | No Reading +FRNT_FAN1_2 | 6Ah | ns | 7.0 | No Reading +""" + + def _read_with_fake_ipmitool(self) -> dict: + reader = FanMetricsReader(exporter_endpoint=None) + fake = mock.Mock(returncode=0, stdout=self.SDR_OUTPUT, stderr="") + with mock.patch("app.sensors.subprocess.run", return_value=fake): + return reader._read_ipmitool() + + def test_only_live_fans_are_returned(self) -> None: + readings = self._read_with_fake_ipmitool() + self.assertEqual( + set(readings), {"CPU1_FAN1", "FRNT_FAN1", "REAR_FAN1", "REAR_FAN2"} + ) + + def test_no_reading_slots_are_skipped(self) -> None: + readings = self._read_with_fake_ipmitool() + for slot in ("FRNT_FAN2", "FRNT_FAN3", "CPU1_FAN1_2", "FRNT_FAN1_2"): + self.assertNotIn(slot, readings, f"{slot} 是未接位,不该出现在结果里") + + def test_rpm_values_are_correct(self) -> None: + readings = self._read_with_fake_ipmitool() + self.assertEqual(readings["FRNT_FAN1"].rpm, 3000.0) + self.assertEqual(readings["REAR_FAN2"].rpm, 3000.0) + self.assertEqual(readings["REAR_FAN1"].rpm, 400.0) + + +class TestSourceAssignments(unittest.TestCase): + """「散热源 → 风扇位」分配模型(2026-09-28 倒置:GPU 是主体)。 + + 这组测试防的是两类事故: + 1. 校验漏洞 —— 两个源抢同一个风扇位 / 分配了不存在的位(数据模型被写脏); + 2. 控制越界 —— 没被分配的风扇位被程序动了(「未分配 = 交回 BMC」的承诺)。 + """ + + GPU_A = "GPU-e60e8f23-b150-6718-8508-3cb00e0d9fc6" + GPU_B = "GPU-e49ed30f-f0f4-dc17-0225-2c1235602b39" + + def _controller(self): + from app.config import AppConfig, FanBinding + from app.controller import FanController + from app.curve import build_curve_from_config + from app.ipmi import IPMIClient + from app.safety import SafetyGuard + from app.sensors import FanMetricsReader, GPUMetricsReader + + # 配置里显式预置两个位(模拟 config.yaml 占位;探测路径另有专项测试) + config = AppConfig( + fans=[FanBinding(slot="FRNT_FAN1"), FanBinding(slot="REAR_FAN2")] + ) + ipmi = mock.MagicMock(spec=IPMIClient) + guard = mock.MagicMock(spec=SafetyGuard) + guard.engaged = False + gpu_reader = mock.MagicMock(spec=GPUMetricsReader) + fan_reader = mock.MagicMock(spec=FanMetricsReader) + # last_source 是实例属性(reader.read() 时才赋值),spec 的 Mock 上没有 + gpu_reader.last_source = "test" + fan_reader.last_source = "test" + curve = build_curve_from_config( + [ + {"temp": 45, "duty": 40}, + {"temp": 75, "duty": 80}, + ] + ) + return FanController(config, ipmi, guard, gpu_reader, fan_reader, curve) + + @staticmethod + def _gpu(uuid: str, temp: float): + """构造一个最小可用的 GPUMetric。""" + from app.sensors import GPUMetric + + return GPUMetric( + uuid=uuid, + index=0, + pci_bus_id="00000000:01:00.0", + model_name="Tesla T10", + temperature=temp, + ) + + # ------------------------------------------------------------ 校验 + + def test_conflicting_slot_is_rejected(self) -> None: + """一个风扇位只能给一个源 —— 冲突必须整体拒绝。""" + c = self._controller() + with self.assertRaises(ValueError): + c.update_assignments( + [ + {"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["FRNT_FAN1"]}, + {"key": "gpu:all", "kind": "gpu_group", "slots": ["FRNT_FAN1", "REAR_FAN2"]}, + ] + ) + # 整体拒绝:不能留下改了一半的状态 + self.assertFalse(c.describe()["bindings_configured"]) + + def test_unknown_slot_is_rejected(self) -> None: + c = self._controller() + with self.assertRaises(ValueError): + c.update_assignments( + [{"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["NOPE"]}] + ) + + def test_gpu_kind_requires_uuid(self) -> None: + c = self._controller() + with self.assertRaises(ValueError): + c.update_assignments([{"key": "gpu:bad", "kind": "gpu", "slots": ["FRNT_FAN1"]}]) + + def test_valid_assignment_builds_owner_index(self) -> None: + c = self._controller() + c.update_assignments( + [ + {"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["FRNT_FAN1"]}, + {"key": f"gpu:{self.GPU_B}", "kind": "gpu", "gpu_uuid": self.GPU_B, "slots": ["REAR_FAN2"]}, + ] + ) + self.assertTrue(c.describe()["bindings_configured"]) + # describe() 里每张卡都能看到自己的风扇位 + rows = {a["key"]: a["slots"] for a in c.describe()["assignments"]} + self.assertEqual(rows[f"gpu:{self.GPU_A}"], ["FRNT_FAN1"]) + self.assertEqual(rows[f"gpu:{self.GPU_B}"], ["REAR_FAN2"]) + + # ------------------------------------------------------------ 控制行为 + + def _apply(self, c, gpus): + """跑一轮曲线,返回 IPMI 收到的 updates 字典。""" + with mock.patch.object(type(c), "_warn_orphan_gpus", lambda self, g: None): + c._apply_curve(gpus) + if c._ipmi.apply.called: + return c._ipmi.apply.call_args[0][0] + return {} + + def test_only_assigned_slots_are_driven(self) -> None: + """控制越界防护:GPU_A 分了 FRNT_FAN1,GPU_B 什么都没分 —— + 再热也只能告警,REAR_FAN2 一根线不能碰。""" + c = self._controller() + c.update_assignments( + [{"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["FRNT_FAN1"]}] + ) + updates = self._apply( + c, + [self._gpu(self.GPU_A, 70.0), self._gpu(self.GPU_B, 88.0)], + ) + self.assertIn("FRNT_FAN1", updates) + self.assertNotIn("REAR_FAN2", updates, "未分配的风扇位绝不能被程序写入") + + def test_each_slot_follows_its_own_gpu(self) -> None: + """两张卡各自驱动自己的风扇位 —— 60°C 的卡不能拉着 80°C 卡的风扇降速。""" + c = self._controller() + c.update_assignments( + [ + {"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["FRNT_FAN1"]}, + {"key": f"gpu:{self.GPU_B}", "kind": "gpu", "gpu_uuid": self.GPU_B, "slots": ["REAR_FAN2"]}, + ] + ) + updates = self._apply( + c, + [self._gpu(self.GPU_A, 60.0), self._gpu(self.GPU_B, 80.0)], + ) + # A 卡 60°C 在第一档(40%);B 卡 80°C 已过 75 折点(80%) + self.assertEqual(updates["FRNT_FAN1"], 40) + self.assertEqual(updates["REAR_FAN2"], 80) + + def test_offline_gpu_releases_its_slots(self) -> None: + """绑定的卡掉卡(读不到温度)→ 它的风扇位交回 BMC 自动(不狂转)。""" + c = self._controller() + c.update_assignments( + [{"key": f"gpu:{self.GPU_B}", "kind": "gpu", "gpu_uuid": self.GPU_B, "slots": ["REAR_FAN2"]}] + ) + updates = self._apply(c, [self._gpu(self.GPU_A, 55.0)]) # B 不在上报里 + self.assertIsNone(updates.get("REAR_FAN2")) + + def test_gpu_group_kind_is_rejected(self) -> None: + """「所有 GPU 最热」合成源已删除(超哥不理解 = 坏选项)—— 传了要拒。""" + c = self._controller() + with self.assertRaises(ValueError): + c.update_assignments( + [{"key": "gpu:all", "kind": "gpu_group", "slots": ["FRNT_FAN1"]}] + ) + + def test_all_slots_listed_even_without_rpm(self) -> None: + """可分配清单 = **全部可控位**(含没有转速读数的),一个都不能少。""" + c = self._controller() + from app.ipmi import FAN_SLOT_INDEX + from app.sensors import FanReading + + # 实测只有 3 个位在转,其余 4 个没接(No Reading)—— 也必须列出 + for slot, rpm in (("FRNT_FAN1", 3000.0), ("REAR_FAN2", 3000.0), ("CPU1_FAN1", 1300.0)): + c._snapshot.fans[slot] = FanReading(slot=slot, rpm=rpm) + + fan_slots = c.describe()["fan_slots"] + self.assertEqual( + [f["slot"] for f in fan_slots], sorted(FAN_SLOT_INDEX) + ) + rpm_of = {f["slot"]: f["rpm"] for f in fan_slots} + self.assertEqual(rpm_of["FRNT_FAN1"], 3000.0) + self.assertIsNone(rpm_of["FRNT_FAN3"], "没读数的位也要列出(rpm=null)") + + # 全部 7 个位都可以正常分配 + c.update_assignments( + [{"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["FRNT_FAN3"]}] + ) + rows = {a["key"]: a["slots"] for a in c.describe()["assignments"]} + self.assertEqual(rows[f"gpu:{self.GPU_A}"], ["FRNT_FAN3"]) + + def test_assignment_to_unmanaged_gpu_is_rejected(self) -> None: + """打通「设置 ↔ 分配」:未纳入管控的卡不能配风扇(明确 400,不留暗状态)。""" + c = self._controller() + c.apply_settings({"control.managed_gpus": [self.GPU_A]}) + with self.assertRaises(ValueError): + c.update_assignments( + [{"key": f"gpu:{self.GPU_B}", "kind": "gpu", "gpu_uuid": self.GPU_B, "slots": ["REAR_FAN2"]}] + ) + + def test_unmanage_clears_assignments(self) -> None: + """设置页取消勾选 → 该卡的分配立即停用并清除(不驱动、不残留)。""" + c = self._controller() + c.update_assignments( + [ + {"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["FRNT_FAN1"]}, + {"key": f"gpu:{self.GPU_B}", "kind": "gpu", "gpu_uuid": self.GPU_B, "slots": ["REAR_FAN2"]}, + ] + ) + # 移出 GPU_B + c.apply_settings({"control.managed_gpus": [self.GPU_A]}) + rows = {a["key"]: a for a in c.export_assignments()} + self.assertEqual(rows[f"gpu:{self.GPU_B}"]["slots"], [], "被移出管控的卡分配应清空") + # FRNT_FAN1 归 GPU_A,REAR_FAN2 已无主 + self.assertEqual(c._slot_owner.get("FRNT_FAN1"), f"gpu:{self.GPU_A}") + self.assertNotIn("REAR_FAN2", c._slot_owner) + # 再给 GPU_B 分配 → 拒绝 + with self.assertRaises(ValueError): + c.update_assignments( + [{"key": f"gpu:{self.GPU_B}", "kind": "gpu", "gpu_uuid": self.GPU_B, "slots": ["REAR_FAN2"]}] + ) + + def test_startup_sanitizes_stale_assignments(self) -> None: + """启动自愈:库里的脏数据(未管控的卡带着分配)只清冲突项,不作废整份。""" + c = self._controller() + c.apply_settings({"control.managed_gpus": [self.GPU_A]}) + stale = [ + {"key": f"gpu:{self.GPU_A}", "kind": "gpu", "gpu_uuid": self.GPU_A, "slots": ["FRNT_FAN1"]}, + # 脏数据:B 未管控却带着分配(模拟上次持久化失败残留) + {"key": f"gpu:{self.GPU_B}", "kind": "gpu", "gpu_uuid": self.GPU_B, "slots": ["REAR_FAN2"]}, + ] + cleaned = c.sanitize_stored_assignments(stale) + self.assertEqual(cleaned[1]["slots"], [], "冲突项应被清空") + # 自愈后整体恢复成功,合法项不受牵连 + c.update_assignments(cleaned) + rows = {a["key"]: a["slots"] for a in c.export_assignments()} + self.assertEqual(rows[f"gpu:{self.GPU_A}"], ["FRNT_FAN1"]) + self.assertEqual(rows[f"gpu:{self.GPU_B}"], []) + + def test_cpu_source_drives_tctl(self) -> None: + c = self._controller() + c.update_assignments([{"key": "cpu", "kind": "cpu", "slots": ["FRNT_FAN1"]}]) + # CPU 温度在 snapshot.cpu_temps 里(模拟 node_exporter 读数) + from app.sensors import CPUCoreTemperature + + c._snapshot.cpu_temps = [CPUCoreTemperature(label="Tctl", celsius=65.0)] + updates = self._apply(c, []) + self.assertEqual(updates["FRNT_FAN1"], 40) # 65°C 低于 75 折点 → 第一档 + + +if __name__ == "__main__": + unittest.main(verbosity=2) From b15e41e8b1c6437ba493147a9dfe3f138b757661 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E8=87=A3=E8=B6=85?= <517024110@qq.com> Date: Mon, 28 Sep 2026 18:56:41 +0800 Subject: [PATCH 05/13] =?UTF-8?q?=E6=96=B0=E5=A2=9E:=20=E6=8E=A7=E5=88=B6?= =?UTF-8?q?=E5=8F=B0=E5=89=8D=E7=AB=AF=EF=BC=88Vue3=20+=20TS=20+=20?= =?UTF-8?q?=E7=8E=B0=E4=BB=A3=20SaaS=20=E9=A3=8E=E5=A4=9A=E9=A1=B5?= =?UTF-8?q?=E9=9D=A2=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 布局: 侧边栏导航(概览 / 风扇控制 / 设置)+ 毛玻璃 sticky 顶栏, 窄屏自动折叠为顶部横条 - 概览页: GPU 指标卡(温度/功耗/显存/分到的风扇)、CPU Tctl 与 BMC 板载温度、Prometheus 历史趋势(15m~3h 分段切换,双 Y 轴) - 风扇控制页: 「GPU → 风扇接口」分配面板(下拉单选、跨源互斥自动 转移、未管控卡置灰禁用)、手动调速滑块(0 = 交回 BMC)、控制曲线图 - 设置页: 管控 GPU 勾选(与分配双向联动)、控制开关/周期、曲线折点 编辑、紧急阈值 —— 全部落 SQLite,保存即生效 - 实时链路: WebSocket 为主、断线自动降级 5s 轮询 + 3s 重连 - 图表统一主题(theme.ts),数字等宽字形,54 项单测覆盖的后端配合 --- frontend/.gitignore | 24 + frontend/.vscode/extensions.json | 3 + frontend/README.md | 5 + frontend/index.html | 12 + frontend/package-lock.json | 1723 ++++++++++++++++++ frontend/package.json | 25 + frontend/public/favicon.svg | 1 + frontend/public/icons.svg | 24 + frontend/src/App.vue | 59 + frontend/src/api.ts | 85 + frontend/src/assets/vite.svg | 1 + frontend/src/components/BindingWizard.vue | 219 +++ frontend/src/components/CurveChart.vue | 110 ++ frontend/src/components/FanPanel.vue | 156 ++ frontend/src/components/GpuCards.vue | 109 ++ frontend/src/components/SettingsPanel.vue | 388 ++++ frontend/src/components/Sidebar.vue | 137 ++ frontend/src/components/TemperaturePanel.vue | 98 + frontend/src/components/TopBar.vue | 140 ++ frontend/src/components/TrendChart.vue | 137 ++ frontend/src/composables/useChart.ts | 46 + frontend/src/composables/useRealtime.ts | 119 ++ frontend/src/main.ts | 5 + frontend/src/pages/DashboardPage.vue | 19 + frontend/src/pages/FansPage.vue | 36 + frontend/src/pages/SettingsPage.vue | 12 + frontend/src/style.css | 67 + frontend/src/theme.ts | 52 + frontend/src/types.ts | 177 ++ frontend/src/utils.ts | 48 + frontend/tsconfig.app.json | 15 + frontend/tsconfig.json | 7 + frontend/tsconfig.node.json | 23 + frontend/vite.config.ts | 33 + 34 files changed, 4115 insertions(+) create mode 100644 frontend/.gitignore create mode 100644 frontend/.vscode/extensions.json create mode 100644 frontend/README.md create mode 100644 frontend/index.html create mode 100644 frontend/package-lock.json create mode 100644 frontend/package.json create mode 100644 frontend/public/favicon.svg create mode 100644 frontend/public/icons.svg create mode 100644 frontend/src/App.vue create mode 100644 frontend/src/api.ts create mode 100644 frontend/src/assets/vite.svg create mode 100644 frontend/src/components/BindingWizard.vue create mode 100644 frontend/src/components/CurveChart.vue create mode 100644 frontend/src/components/FanPanel.vue create mode 100644 frontend/src/components/GpuCards.vue create mode 100644 frontend/src/components/SettingsPanel.vue create mode 100644 frontend/src/components/Sidebar.vue create mode 100644 frontend/src/components/TemperaturePanel.vue create mode 100644 frontend/src/components/TopBar.vue create mode 100644 frontend/src/components/TrendChart.vue create mode 100644 frontend/src/composables/useChart.ts create mode 100644 frontend/src/composables/useRealtime.ts create mode 100644 frontend/src/main.ts create mode 100644 frontend/src/pages/DashboardPage.vue create mode 100644 frontend/src/pages/FansPage.vue create mode 100644 frontend/src/pages/SettingsPage.vue create mode 100644 frontend/src/style.css create mode 100644 frontend/src/theme.ts create mode 100644 frontend/src/types.ts create mode 100644 frontend/src/utils.ts create mode 100644 frontend/tsconfig.app.json create mode 100644 frontend/tsconfig.json create mode 100644 frontend/tsconfig.node.json create mode 100644 frontend/vite.config.ts diff --git a/frontend/.gitignore b/frontend/.gitignore new file mode 100644 index 0000000..a547bf3 --- /dev/null +++ b/frontend/.gitignore @@ -0,0 +1,24 @@ +# Logs +logs +*.log +npm-debug.log* +yarn-debug.log* +yarn-error.log* +pnpm-debug.log* +lerna-debug.log* + +node_modules +dist +dist-ssr +*.local + +# Editor directories and files +.vscode/* +!.vscode/extensions.json +.idea +.DS_Store +*.suo +*.ntvs* +*.njsproj +*.sln +*.sw? diff --git a/frontend/.vscode/extensions.json b/frontend/.vscode/extensions.json new file mode 100644 index 0000000..a7cea0b --- /dev/null +++ b/frontend/.vscode/extensions.json @@ -0,0 +1,3 @@ +{ + "recommendations": ["Vue.volar"] +} diff --git a/frontend/README.md b/frontend/README.md new file mode 100644 index 0000000..33895ab --- /dev/null +++ b/frontend/README.md @@ -0,0 +1,5 @@ +# Vue 3 + TypeScript + Vite + +This template should help get you started developing with Vue 3 and TypeScript in Vite. The template uses Vue 3 ` + + diff --git a/frontend/package-lock.json b/frontend/package-lock.json new file mode 100644 index 0000000..0ad5952 --- /dev/null +++ b/frontend/package-lock.json @@ -0,0 +1,1723 @@ +{ + "name": "frontend", + "version": "0.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "frontend", + "version": "0.0.0", + "dependencies": { + "@tailwindcss/vite": "^4.3.3", + "echarts": "^6.1.0", + "tailwindcss": "^4.3.3", + "vue": "^3.5.42" + }, + "devDependencies": { + "@types/node": "^24.13.3", + "@vitejs/plugin-vue": "^6.0.8", + "@vue/tsconfig": "^0.9.1", + "typescript": "~6.0.2", + "vite": "^8.3.0", + "vue-tsc": "^3.3.11" + } + }, + "node_modules/@babel/helper-string-parser": { + "version": "7.29.7", + "resolved": "https://mirrors.cloud.tencent.com/npm/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz", + "integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==", + "license": "MIT", + "engines": { + "node": ">=6.9.0" + } + }, + "node_modules/@babel/helper-validator-identifier": { + "version": "7.29.7", + "resolved": "https://mirrors.cloud.tencent.com/npm/@babel/helper-validator-identifier/-/helper-validator-identifier-7.29.7.tgz", + "integrity": "sha512-qehxGkRj55h/ff8EMaJ+cYhyaKlHIxqYDn682wQD7RNp9UujOQsHog2uS0r2vzr4pW+sXf90NeeayjcNaX3fFg==", + "license": "MIT", + "engines": { + "node": ">=6.9.0" + } + }, + "node_modules/@babel/parser": { + "version": "7.29.9", + "resolved": "https://mirrors.cloud.tencent.com/npm/@babel/parser/-/parser-7.29.9.tgz", + "integrity": "sha512-CjXrNHTnvqBVqHgdBysY3vk2T8tpJHb5/RMeHJBTyVa9xgugCB0CJTx/3oO8RV2QRQP391RWpB7D6hLjm8V9uA==", + "dependencies": { + "@babel/types": "^7.29.8" + }, + "bin": { + "parser": "bin/babel-parser.js" + }, + "engines": { + "node": ">=6.0.0" + } + }, + "node_modules/@babel/types": { + "version": "7.29.8", + "resolved": "https://mirrors.cloud.tencent.com/npm/@babel/types/-/types-7.29.8.tgz", + "integrity": "sha512-Vj1jF3cPfxg7OAfoI7QnVKLoILlm2JF9pnVHrX8qx7AHMiYWT+NDAA7jChlNgRS4WTLc/fD1lXLmPixluj+3Gg==", + "license": "MIT", + "dependencies": { + "@babel/helper-string-parser": "^7.29.7", + "@babel/helper-validator-identifier": "^7.29.7" + }, + "engines": { + "node": ">=6.9.0" + } + }, + "node_modules/@jridgewell/gen-mapping": { + "version": "0.3.13", + "resolved": "https://mirrors.cloud.tencent.com/npm/@jridgewell/gen-mapping/-/gen-mapping-0.3.13.tgz", + "integrity": "sha512-2kkt/7niJ6MgEPxF0bYdQ6etZaA+fQvDcLKckhy1yIQOzaoKjBBjSj63/aLVjYE3qhRt5dvM+uUyfCg6UKCBbA==", + "license": "MIT", + "dependencies": { + "@jridgewell/sourcemap-codec": "^1.5.0", + "@jridgewell/trace-mapping": "^0.3.24" + } + }, + "node_modules/@jridgewell/remapping": { + "version": "2.3.5", + "resolved": "https://mirrors.cloud.tencent.com/npm/@jridgewell/remapping/-/remapping-2.3.5.tgz", + "integrity": "sha512-LI9u/+laYG4Ds1TDKSJW2YPrIlcVYOwi2fUC6xB43lueCjgxV4lffOCZCtYFiH6TNOX+tQKXx97T4IKHbhyHEQ==", + "license": "MIT", + "dependencies": { + "@jridgewell/gen-mapping": "^0.3.5", + "@jridgewell/trace-mapping": "^0.3.24" + } + }, + "node_modules/@jridgewell/resolve-uri": { + "version": "3.1.2", + "resolved": "https://mirrors.cloud.tencent.com/npm/@jridgewell/resolve-uri/-/resolve-uri-3.1.2.tgz", + "integrity": "sha512-bRISgCIjP20/tbWSPWMEi54QVPRZExkuD9lJL+UIxUKtwVJA8wW1Trb1jMs1RFXo1CBTNZ/5hpC9QvmKWdopKw==", + "license": "MIT", + "engines": { + "node": ">=6.0.0" + } + }, + "node_modules/@jridgewell/sourcemap-codec": { + "version": "1.6.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/@jridgewell/sourcemap-codec/-/sourcemap-codec-1.6.0.tgz", + "integrity": "sha512-T7jf+5zgsZHwNJ4lvQ7/aezbyk0nNX+zJVWpmHA7VYsEx7a7qr5Rg5IbtJFqkgze5Y2sruq1RUY8Q837Od7iFw==", + "license": "MIT" + }, + "node_modules/@jridgewell/trace-mapping": { + "version": "0.3.31", + "resolved": "https://mirrors.cloud.tencent.com/npm/@jridgewell/trace-mapping/-/trace-mapping-0.3.31.tgz", + "integrity": "sha512-zzNR+SdQSDJzc8joaeP8QQoCQr8NuYx2dIIytl1QeBEZHJ9uW6hebsrYgbz8hJwUQao3TWCMtmfV8Nu1twOLAw==", + "license": "MIT", + "dependencies": { + "@jridgewell/resolve-uri": "^3.1.0", + "@jridgewell/sourcemap-codec": "^1.4.14" + } + }, + "node_modules/@oxc-project/types": { + "version": "0.151.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/@oxc-project/types/-/types-0.151.0.tgz", + "integrity": "sha512-J1yXrIlNDZVzE3ada310xeAw7nH8yCAyLPuUIsjKatFPmfn5bS1oW+cM+QsGOtVWd5nhSpbwZWx/rue+r5Z+PA==", + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/oxc-project" + } + }, + "node_modules/@rolldown/binding-android-arm-eabi": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-android-arm-eabi/-/binding-android-arm-eabi-1.2.11.tgz", + "integrity": "sha512-A5kXfGKvKWWZE0TtPrfsvT+q4Y5d1QG8gGUzpYjGydM+fARM9MuX90PrXYXe0XbsDVgyxxNzHo6giCj90bsFNw==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-android-arm64": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-android-arm64/-/binding-android-arm64-1.2.11.tgz", + "integrity": "sha512-z6cTycz+iJ4PVkuL4HHW4DfTfoeU/2nqYYuSOrTmH7yHK5Y0LCOnA03V4ZNxavyVaU1oOqUgIg2klN/s+USGOA==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-darwin-arm64": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.2.11.tgz", + "integrity": "sha512-jShvqNtP6vDC6/A5JOAzbVV+DkgHqhl/ScVCJEbt+TUY6QYz7YnXcrg3sLtFBniro0f/Ld50ZwCWA6f7KYD1nQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-darwin-x64": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.2.11.tgz", + "integrity": "sha512-f2i2xiNWq1Z1l2++q2fuhZRdLAT3aqxD6vRNm1RAxpUoBcdqNB3C0s1Bt+K+PbEx2F5F4gQp6hqKkphCY/xF9w==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-freebsd-x64": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.2.11.tgz", + "integrity": "sha512-4Ir5FSOKIAMr4r0kExpt1s3bMgzJU3rA45AYOHtQpls0oNeqcYBKrWMlckrYH4KCfGLfkfn1tN1dmZPMVsdXow==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-linux-arm-gnueabihf": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.2.11.tgz", + "integrity": "sha512-/gnRDM+39BROzAN/k1OZjDPnDMcZxB/0EUxKjONO5yVkNEvlsoMDrxGNKgZi/ttFriS2gwlDNzB65pvNbFOXIQ==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-linux-arm64-gnu": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.2.11.tgz", + "integrity": "sha512-PFaK8HwvAHbaKbBcDNQihjMKYvFnA5hiENx/l5tphTDz1E0WFp32l0A7aq7lyUwGsRw/xSrNIy/gIK4thrSCrw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-linux-arm64-musl": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.2.11.tgz", + "integrity": "sha512-AskzJUIKRLPxkruR1wLKewGbOw+EYfU/9lOrBFj4AFrEA8hPpKFnODWNu2WLaNs0QNkEb9QIJufmVZZIL/bJlg==", + "cpu": [ + "arm64" + ], + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-linux-ppc64-gnu": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.2.11.tgz", + "integrity": "sha512-qlUGAheh2yh8afH7QBgx0PrRHN85hKnNd78x8MeMhXivuevgd8vgf6/CstOzmNKY/lLTHvNTrPy98cLnAugzJw==", + "cpu": [ + "ppc64" + ], + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-linux-s390x-gnu": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.2.11.tgz", + "integrity": "sha512-secpEad+0vCbSfn8upFySkDskv+bGPk3THSDS9Y89yc4rb4kzqHp8Dmyd9BkQW4SnhNXBZCl/6CrO//hZahNJQ==", + "cpu": [ + "s390x" + ], + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-linux-x64-gnu": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.2.11.tgz", + "integrity": "sha512-mOVBT3dPpkWm8XBWPmU4bf+U6dYDLeMo/9ojUmis4N0L5uu10qra5vOyngZ7/PSdoE4G9KvRt4bloRxNjLas7A==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-linux-x64-musl": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.2.11.tgz", + "integrity": "sha512-Is78i9A8Ui4SqcxUwFJ9uMmjDn58IbVTjFWYdQestFEgeuEmHMLGNriXnVJKkwG2YiZjw8cP0zCTyDMdDGtOOg==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-openharmony-arm64": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.2.11.tgz", + "integrity": "sha512-dUCXneZ87INUMyQ0D+C0HrEBNUPNXHaPmU5GTjyKTJEiussw9Kaj5Ln8UztPe4epV/ffvgNBEadksdYhmW6xJA==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "openharmony" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-win32-arm64-msvc": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.2.11.tgz", + "integrity": "sha512-jByxb6qfd+bH1xUd0qnfFnb17i9sWBPY2tOavJ0l3tdr3OTu+Kvtm8cd/JV5nFt657b1VqGltxg9olOEfofXWw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/binding-win32-x64-msvc": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.2.11.tgz", + "integrity": "sha512-/PzKqzAJ03i19oy2ItPvyvaVjOjBCNnfaJs8yvUdGBKmiESgnrJSQ2awd81QzFbbnAmu7YO9ZnJrDCb9VSJPRA==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@rolldown/pluginutils": { + "version": "1.0.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/@rolldown/pluginutils/-/pluginutils-1.0.1.tgz", + "integrity": "sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw==" + }, + "node_modules/@tailwindcss/node": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/node/-/node-4.3.3.tgz", + "integrity": "sha512-/T8IKEsf9VTU6tLjgC7+sv2mOPtQxzE2jMw7u4Tt40Tx+QSZxpzh95/H6cMKoja9XuW7iMdLJYBB0o9G1CaAgg==", + "license": "MIT", + "dependencies": { + "@jridgewell/remapping": "^2.3.5", + "enhanced-resolve": "^5.24.1", + "jiti": "^2.7.0", + "lightningcss": "1.32.0", + "magic-string": "^0.30.21", + "source-map-js": "^1.2.1", + "tailwindcss": "4.3.3" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss/-/lightningcss-1.32.0.tgz", + "integrity": "sha512-NXYBzinNrblfraPGyrbPoD19C1h9lfI/1mzgWYvXUTe414Gz/X1FD2XBZSZM7rRTrMA8JL3OtAaGifrIKhQ5yQ==", + "license": "MPL-2.0", + "dependencies": { + "detect-libc": "^2.0.3" + }, + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + }, + "optionalDependencies": { + "lightningcss-android-arm64": "1.32.0", + "lightningcss-darwin-arm64": "1.32.0", + "lightningcss-darwin-x64": "1.32.0", + "lightningcss-freebsd-x64": "1.32.0", + "lightningcss-linux-arm-gnueabihf": "1.32.0", + "lightningcss-linux-arm64-gnu": "1.32.0", + "lightningcss-linux-arm64-musl": "1.32.0", + "lightningcss-linux-x64-gnu": "1.32.0", + "lightningcss-linux-x64-musl": "1.32.0", + "lightningcss-win32-arm64-msvc": "1.32.0", + "lightningcss-win32-x64-msvc": "1.32.0" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-android-arm64": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-android-arm64/-/lightningcss-android-arm64-1.32.0.tgz", + "integrity": "sha512-YK7/ClTt4kAK0vo6w3X+Pnm0D2cf2vPHbhOXdoNti1Ga0al1P4TBZhwjATvjNwLEBCnKvjJc2jQgHXH0NEwlAg==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-darwin-arm64": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-darwin-arm64/-/lightningcss-darwin-arm64-1.32.0.tgz", + "integrity": "sha512-RzeG9Ju5bag2Bv1/lwlVJvBE3q6TtXskdZLLCyfg5pt+HLz9BqlICO7LZM7VHNTTn/5PRhHFBSjk5lc4cmscPQ==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-darwin-x64": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-darwin-x64/-/lightningcss-darwin-x64-1.32.0.tgz", + "integrity": "sha512-U+QsBp2m/s2wqpUYT/6wnlagdZbtZdndSmut/NJqlCcMLTWp5muCrID+K5UJ6jqD2BFshejCYXniPDbNh73V8w==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-freebsd-x64": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-freebsd-x64/-/lightningcss-freebsd-x64-1.32.0.tgz", + "integrity": "sha512-JCTigedEksZk3tHTTthnMdVfGf61Fky8Ji2E4YjUTEQX14xiy/lTzXnu1vwiZe3bYe0q+SpsSH/CTeDXK6WHig==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-linux-arm-gnueabihf": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-arm-gnueabihf/-/lightningcss-linux-arm-gnueabihf-1.32.0.tgz", + "integrity": "sha512-x6rnnpRa2GL0zQOkt6rts3YDPzduLpWvwAF6EMhXFVZXD4tPrBkEFqzGowzCsIWsPjqSK+tyNEODUBXeeVHSkw==", + "cpu": [ + "arm" + ], + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-linux-arm64-gnu": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-arm64-gnu/-/lightningcss-linux-arm64-gnu-1.32.0.tgz", + "integrity": "sha512-0nnMyoyOLRJXfbMOilaSRcLH3Jw5z9HDNGfT/gwCPgaDjnx0i8w7vBzFLFR1f6CMLKF8gVbebmkUN3fa/kQJpQ==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-linux-arm64-musl": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-arm64-musl/-/lightningcss-linux-arm64-musl-1.32.0.tgz", + "integrity": "sha512-UpQkoenr4UJEzgVIYpI80lDFvRmPVg6oqboNHfoH4CQIfNA+HOrZ7Mo7KZP02dC6LjghPQJeBsvXhJod/wnIBg==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-linux-x64-gnu": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-x64-gnu/-/lightningcss-linux-x64-gnu-1.32.0.tgz", + "integrity": "sha512-V7Qr52IhZmdKPVr+Vtw8o+WLsQJYCTd8loIfpDaMRWGUZfBOYEJeyJIkqGIDMZPwPx24pUMfwSxxI8phr/MbOA==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-linux-x64-musl": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-x64-musl/-/lightningcss-linux-x64-musl-1.32.0.tgz", + "integrity": "sha512-bYcLp+Vb0awsiXg/80uCRezCYHNg1/l3mt0gzHnWV9XP1W5sKa5/TCdGWaR/zBM2PeF/HbsQv/j2URNOiVuxWg==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-win32-arm64-msvc": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-win32-arm64-msvc/-/lightningcss-win32-arm64-msvc-1.32.0.tgz", + "integrity": "sha512-8SbC8BR40pS6baCM8sbtYDSwEVQd4JlFTOlaD3gWGHfThTcABnNDBda6eTZeqbofalIJhFx0qKzgHJmcPTnGdw==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/node/node_modules/lightningcss-win32-x64-msvc": { + "version": "1.32.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-win32-x64-msvc/-/lightningcss-win32-x64-msvc-1.32.0.tgz", + "integrity": "sha512-Amq9B/SoZYdDi1kFrojnoqPLxYhQ4Wo5XiL8EVJrVsB8ARoC1PWW6VGtT0WKCemjy8aC+louJnjS7U18x3b06Q==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/@tailwindcss/oxide": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide/-/oxide-4.3.3.tgz", + "integrity": "sha512-krXjAikiaFSPaK/FkAQT5UTx3VormQaiZ5hBFlJZ9UFQGB/rwg1MZIhHAG9smMQRTdyJxP6Qt5MwMtdyU5FWrA==", + "engines": { + "node": ">= 20" + }, + "optionalDependencies": { + "@tailwindcss/oxide-android-arm64": "4.3.3", + "@tailwindcss/oxide-darwin-arm64": "4.3.3", + "@tailwindcss/oxide-darwin-x64": "4.3.3", + "@tailwindcss/oxide-freebsd-x64": "4.3.3", + "@tailwindcss/oxide-linux-arm-gnueabihf": "4.3.3", + "@tailwindcss/oxide-linux-arm64-gnu": "4.3.3", + "@tailwindcss/oxide-linux-arm64-musl": "4.3.3", + "@tailwindcss/oxide-linux-x64-gnu": "4.3.3", + "@tailwindcss/oxide-linux-x64-musl": "4.3.3", + "@tailwindcss/oxide-wasm32-wasi": "4.3.3", + "@tailwindcss/oxide-win32-arm64-msvc": "4.3.3", + "@tailwindcss/oxide-win32-x64-msvc": "4.3.3" + } + }, + "node_modules/@tailwindcss/oxide-android-arm64": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-android-arm64/-/oxide-android-arm64-4.3.3.tgz", + "integrity": "sha512-Y85A2gmPSkl5Ve5qR86GL4HT509cFqQh1aes9p3sSkyTPwt0Pppf3GkwGe4JPACcRYjgJIEhQgM6dBClnr0NYw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-darwin-arm64": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-darwin-arm64/-/oxide-darwin-arm64-4.3.3.tgz", + "integrity": "sha512-BiaWatpBcERQFDlOjRDpIVXuFK5PJez5SA4JMg6VYZdBYU+qKfV/vqjcIs+IYmtitf1xYQZTwXvU/8y4lfZUGw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-darwin-x64": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-darwin-x64/-/oxide-darwin-x64-4.3.3.tgz", + "integrity": "sha512-fAeUqfV5ndhxRwai8cXGzdLvul9utWOmeTkv69unv4ZXixjn61Z+p9lCWdwOwA3TYboG3BwdVuN/RDjhBRl0mw==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-freebsd-x64": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-freebsd-x64/-/oxide-freebsd-x64-4.3.3.tgz", + "integrity": "sha512-iyf5bV6+wnAlflVeEy7R25dupxTNECZN5QMI0qNT6eT+EgaGdZcKhGkr5SdoaWiLJ3spLqIY9VCeSGrwmtg4kw==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-linux-arm-gnueabihf": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-linux-arm-gnueabihf/-/oxide-linux-arm-gnueabihf-4.3.3.tgz", + "integrity": "sha512-aAYUprJAJQWWbRrPvtjdroZ56Md+JM8pMiopS6xGEwDfLhqj+2ver2p4nU4Mb3CRqcMmNBjo8KkUgcxhkzVQGQ==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-linux-arm64-gnu": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-linux-arm64-gnu/-/oxide-linux-arm64-gnu-4.3.3.tgz", + "integrity": "sha512-nDxldcEENOxZRzC2uu9jrutZdAAQtb+8WWDCSnWL1zvBk1+FN+x6MtDViPB5AJMfttVCUhehGWus3XBPgatM/w==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-linux-arm64-musl": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-linux-arm64-musl/-/oxide-linux-arm64-musl-4.3.3.tgz", + "integrity": "sha512-Md44bD6veX/PC5iyF8cDVnw4HBIANZepRZZ7a8DQOvkfo5WUBwcp6iAuCUz23u+4SUkhJlD3eL7hNdW8ezd/kA==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-linux-x64-gnu": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-linux-x64-gnu/-/oxide-linux-x64-gnu-4.3.3.tgz", + "integrity": "sha512-tx7us1muwOKAKWao2v/GaafFeQboE6aj88vC6ziN2NCGcRm8gWUhwjzg+YdVB1e4boAtdtma4L43onunI6NS4w==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-linux-x64-musl": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-linux-x64-musl/-/oxide-linux-x64-musl-4.3.3.tgz", + "integrity": "sha512-SJxX60smvHgasZoBy11dX6YRjXJFovwWBoedhbQPOBzgFWBHGB+TVPWB9BxzR7TTxU8FQZAI2AyiNCMzFm8Img==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-wasm32-wasi": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-wasm32-wasi/-/oxide-wasm32-wasi-4.3.3.tgz", + "integrity": "sha512-jx1+rPhY/5Ympkktd656HBWEBLxP7dH06losBLjjf5vgCODXvi9KhtftWcMIwTFIDqBr7cRnQkdLnAG+IOlGvQ==", + "bundleDependencies": [ + "@napi-rs/wasm-runtime", + "@emnapi/core", + "@emnapi/runtime", + "@tybys/wasm-util", + "@emnapi/wasi-threads", + "tslib" + ], + "cpu": [ + "wasm32" + ], + "optional": true, + "dependencies": { + "@emnapi/core": "^1.11.1", + "@emnapi/runtime": "^1.11.1", + "@emnapi/wasi-threads": "^1.2.2", + "@napi-rs/wasm-runtime": "^1.1.4", + "@tybys/wasm-util": "^0.10.2", + "tslib": "^2.8.1" + }, + "engines": { + "node": ">=14.0.0" + } + }, + "node_modules/@tailwindcss/oxide-win32-arm64-msvc": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-win32-arm64-msvc/-/oxide-win32-arm64-msvc-4.3.3.tgz", + "integrity": "sha512-3rc292Ca2ceK6Ulcc/bAVnTs/3nDtoPhyEKlgPv+yQJQi/JS/AMJlqzxvlDacL1nekbrcf6bTqp/jV4qgnPxNQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/oxide-win32-x64-msvc": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/oxide-win32-x64-msvc/-/oxide-win32-x64-msvc-4.3.3.tgz", + "integrity": "sha512-yJ0pwIVc/nYeGoV02WtsN8KYyLQv7kyI2wDnkezyJlGGjkd4QLwDGAwl47YpPJeuI0M0ObaXGSPjvWDPeTPggw==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 20" + } + }, + "node_modules/@tailwindcss/vite": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/@tailwindcss/vite/-/vite-4.3.3.tgz", + "integrity": "sha512-yYU8cogLeSh/ms2jh8Fj7jaba/EWa7Ja6GoUqYZaraEuCI5YS6ms6ObZgjjedm+jm6XZjdNRWBpPP6Z86oOxcw==", + "license": "MIT", + "dependencies": { + "@tailwindcss/node": "4.3.3", + "@tailwindcss/oxide": "4.3.3", + "tailwindcss": "4.3.3" + }, + "peerDependencies": { + "vite": "^5.2.0 || ^6 || ^7 || ^8" + } + }, + "node_modules/@types/node": { + "version": "24.19.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/@types/node/-/node-24.19.0.tgz", + "integrity": "sha512-zY+5tKxXdhGh1PYI0ac+7juvEu4OI6vWtVVoj5i2m42jxAY1U+zHGt6QCyOFwykdP62sM3MJ9stoYYUw5aCWew==", + "devOptional": true, + "license": "MIT", + "dependencies": { + "undici-types": ">=7.24.0 <7.24.7" + } + }, + "node_modules/@vitejs/plugin-vue": { + "version": "6.0.9", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vitejs/plugin-vue/-/plugin-vue-6.0.9.tgz", + "integrity": "sha512-rD/MORlhaZMlXWW0rEn4FB4wMinWC0z/D7Ye160S5+Rs1mC8ZDAdcDp4SjKfUhG5t8q1ktcPVw4xuTkPRlzrSA==", + "dev": true, + "dependencies": { + "@rolldown/pluginutils": "^1.0.1" + }, + "engines": { + "node": "^20.19.0 || >=22.12.0" + }, + "peerDependencies": { + "vite": "^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0", + "vue": "^3.2.25" + } + }, + "node_modules/@volar/language-core": { + "version": "2.4.28", + "resolved": "https://mirrors.cloud.tencent.com/npm/@volar/language-core/-/language-core-2.4.28.tgz", + "integrity": "sha512-w4qhIJ8ZSitgLAkVay6AbcnC7gP3glYM3fYwKV3srj8m494E3xtrCv6E+bWviiK/8hs6e6t1ij1s2Endql7vzQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "@volar/source-map": "2.4.28" + } + }, + "node_modules/@volar/source-map": { + "version": "2.4.28", + "resolved": "https://mirrors.cloud.tencent.com/npm/@volar/source-map/-/source-map-2.4.28.tgz", + "integrity": "sha512-yX2BDBqJkRXfKw8my8VarTyjv48QwxdJtvRgUpNE5erCsgEUdI2DsLbpa+rOQVAJYshY99szEcRDmyHbF10ggQ==", + "dev": true, + "license": "MIT" + }, + "node_modules/@volar/typescript": { + "version": "2.4.28", + "resolved": "https://mirrors.cloud.tencent.com/npm/@volar/typescript/-/typescript-2.4.28.tgz", + "integrity": "sha512-Ja6yvWrbis2QtN4ClAKreeUZPVYMARDYZl9LMEv1iQ1QdepB6wn0jTRxA9MftYmYa4DQ4k/DaSZpFPUfxl8giw==", + "dev": true, + "license": "MIT", + "dependencies": { + "@volar/language-core": "2.4.28", + "path-browserify": "^1.0.1", + "vscode-uri": "^3.0.8" + } + }, + "node_modules/@vue/compiler-core": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/compiler-core/-/compiler-core-3.5.43.tgz", + "integrity": "sha512-zdiLhnbe1QQqgDT8xZMpNmyqZ3qlI+/Q/FHQco57Kwl/b05HhCzN6eVGN9QU9rbga4CrS0H5SYY8VGHZCt/1Hg==", + "license": "MIT", + "dependencies": { + "@babel/parser": "^7.29.8", + "@vue/shared": "3.5.43", + "entities": "^7.0.1", + "estree-walker": "^2.0.2", + "source-map-js": "^1.2.1" + } + }, + "node_modules/@vue/compiler-dom": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/compiler-dom/-/compiler-dom-3.5.43.tgz", + "integrity": "sha512-PEZoAk3NQmsn/ejMzSOCyTYqwGqczrWm70PuhBKjjv1+TCoQAaO/zOqNwjV+honlNstT5ILxtc+8r8UUfj+iEQ==", + "license": "MIT", + "dependencies": { + "@vue/compiler-core": "3.5.43", + "@vue/shared": "3.5.43" + } + }, + "node_modules/@vue/compiler-sfc": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/compiler-sfc/-/compiler-sfc-3.5.43.tgz", + "integrity": "sha512-FCbrG3XNCRl+js3huuKx4IVHBLTvMkJhVepjbxSPu1gn4yWLaYtGNQdjJGZaMytXB6qb76qQDDmDSLy/vkmleQ==", + "license": "MIT", + "dependencies": { + "@babel/parser": "^7.29.8", + "@vue/compiler-core": "3.5.43", + "@vue/compiler-dom": "3.5.43", + "@vue/compiler-ssr": "3.5.43", + "@vue/shared": "3.5.43", + "estree-walker": "^2.0.2", + "magic-string": "^0.30.21", + "postcss": "^8.5.28", + "source-map-js": "^1.2.1" + } + }, + "node_modules/@vue/compiler-ssr": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/compiler-ssr/-/compiler-ssr-3.5.43.tgz", + "integrity": "sha512-GF62orf7KiJX9RqrHNGrYBudsQGD0OhJ5nDs90O8UiDuS40+YMYomiXu6w6EuvtXuRDc3MSNis3EaaSKAVSWpg==", + "license": "MIT", + "dependencies": { + "@vue/compiler-dom": "3.5.43", + "@vue/shared": "3.5.43" + } + }, + "node_modules/@vue/language-core": { + "version": "3.3.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/language-core/-/language-core-3.3.11.tgz", + "integrity": "sha512-QJmpliwAVpC/OxubIByPAhNzsQPRc8/gxlN2qnVzVfIMjMDz/9RnXRFoetjz5yEgXVXyp4LqhXq3V53PjmNzFw==", + "dev": true, + "license": "MIT", + "dependencies": { + "@volar/language-core": "2.4.28", + "@vue/compiler-dom": "^3.5.0", + "@vue/shared": "^3.5.0", + "alien-signals": "^3.2.1", + "muggle-string": "^0.4.1", + "path-browserify": "^1.0.1", + "picomatch": "^4.0.4" + } + }, + "node_modules/@vue/reactivity": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/reactivity/-/reactivity-3.5.43.tgz", + "integrity": "sha512-G/c9GyOZNI2jVaaS6OX1EF1SSFSv7H0ERqNTl4+DTFMlZmB5eVAB53aLQNam/7NL2NPtaDD7RdVrzf8uJzMuOA==", + "license": "MIT", + "dependencies": { + "@vue/shared": "3.5.43" + } + }, + "node_modules/@vue/runtime-core": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/runtime-core/-/runtime-core-3.5.43.tgz", + "integrity": "sha512-hU6U6VnVhBGQDpvlnnDlIB8ZGJBiOcgk2lh/0InltHiz3D8oSkluvuvY+do1G2H3+udeKFsmaBlgVYP7gXQzEw==", + "dependencies": { + "@vue/reactivity": "3.5.43", + "@vue/shared": "3.5.43" + } + }, + "node_modules/@vue/runtime-dom": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/runtime-dom/-/runtime-dom-3.5.43.tgz", + "integrity": "sha512-Bb2Jc0YjjJdMt1SJmb9b2L/IWd3I8lIT9x9eS/xvvP9CiVgna0ffua74xKRmt4/uSJ+0r4iN8ex1jrqkhQGWQw==", + "license": "MIT", + "dependencies": { + "@vue/reactivity": "3.5.43", + "@vue/runtime-core": "3.5.43", + "@vue/shared": "3.5.43", + "csstype": "^3.2.3" + } + }, + "node_modules/@vue/server-renderer": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/server-renderer/-/server-renderer-3.5.43.tgz", + "integrity": "sha512-l2Ygjv9NehV94PSBxNWsAHC0j/eIIKbn92mBuWXAPGLnn6HfJ6MH5ubsd+Nk0YoZ5FRuxWI1P2VSoh+dbPQhCQ==", + "dependencies": { + "@vue/compiler-ssr": "3.5.43", + "@vue/runtime-dom": "3.5.43", + "@vue/shared": "3.5.43" + } + }, + "node_modules/@vue/shared": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/shared/-/shared-3.5.43.tgz", + "integrity": "sha512-uksS7YGMR5NZyr4JNq0Rp+QyLns0ueaz20KwzIPW9R0LH1Vnt4E+XUM29PNseEbf1www2gOuhuDi5AKOIXag9Q==", + "license": "MIT" + }, + "node_modules/@vue/tsconfig": { + "version": "0.9.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/@vue/tsconfig/-/tsconfig-0.9.1.tgz", + "integrity": "sha512-buvjm+9NzLCJL29KY1j1991YYJ5e6275OiK+G4jtmfIb+z4POywbdm0wXusT9adVWqe0xqg70TbI7+mRx4uU9w==", + "dev": true, + "license": "MIT", + "peerDependencies": { + "typescript": ">= 5.8", + "vue": "^3.4.0" + }, + "peerDependenciesMeta": { + "typescript": { + "optional": true + }, + "vue": { + "optional": true + } + } + }, + "node_modules/alien-signals": { + "version": "3.2.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/alien-signals/-/alien-signals-3.2.1.tgz", + "integrity": "sha512-I8FjmltrfnDFoZedi5CG8DghVYNhzb/Ijluz7tCSJH0xpd0484Kowhbb1XDYOxfJpU1p5wnM2X54dA+IfGyD1g==", + "dev": true, + "license": "MIT" + }, + "node_modules/csstype": { + "version": "3.2.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/csstype/-/csstype-3.2.3.tgz", + "integrity": "sha512-z1HGKcYy2xA8AGQfwrn0PAy+PB7X/GSj3UVJW9qKyn43xWa+gl5nXmU4qqLMRzWVLFC8KusUX8T/0kCiOYpAIQ==", + "license": "MIT" + }, + "node_modules/detect-libc": { + "version": "2.1.2", + "resolved": "https://mirrors.cloud.tencent.com/npm/detect-libc/-/detect-libc-2.1.2.tgz", + "integrity": "sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==", + "license": "Apache-2.0", + "engines": { + "node": ">=8" + } + }, + "node_modules/echarts": { + "version": "6.1.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/echarts/-/echarts-6.1.0.tgz", + "integrity": "sha512-q0yaFPggC9FUdsWH4blavRWFmxdrIodbkoKNAjJudAI6CA9gNPxHtV2RcZNEepZVlk4yvBYkOkbk6HIVpIyHZA==", + "license": "Apache-2.0", + "dependencies": { + "tslib": "2.3.0", + "zrender": "6.1.0" + } + }, + "node_modules/enhanced-resolve": { + "version": "5.25.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/enhanced-resolve/-/enhanced-resolve-5.25.1.tgz", + "integrity": "sha512-nGXts5znJzmWPu+mIE9izCOzdg63oJca2mDzGWWTth7sr4aCToKcoyFVBQwN75Ij5Pf6p510EwkTqViTRzDV+w==", + "dependencies": { + "graceful-fs": "^4.2.4", + "tapable": "^2.3.3" + }, + "engines": { + "node": ">=10.13.0" + } + }, + "node_modules/entities": { + "version": "7.0.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/entities/-/entities-7.0.1.tgz", + "integrity": "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA==", + "license": "BSD-2-Clause", + "engines": { + "node": ">=0.12" + }, + "funding": { + "url": "https://github.com/fb55/entities?sponsor=1" + } + }, + "node_modules/estree-walker": { + "version": "2.0.2", + "resolved": "https://mirrors.cloud.tencent.com/npm/estree-walker/-/estree-walker-2.0.2.tgz", + "integrity": "sha512-Rfkk/Mp/DL7JVje3u18FxFujQlTNR2q6QfMSMB7AvCBx91NGj/ba3kCfza0f6dVDbw7YlRf/nDrn7pQrCCyQ/w==", + "license": "MIT" + }, + "node_modules/fdir": { + "version": "6.5.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/fdir/-/fdir-6.5.0.tgz", + "integrity": "sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg==", + "license": "MIT", + "engines": { + "node": ">=12.0.0" + }, + "peerDependencies": { + "picomatch": "^3 || ^4" + }, + "peerDependenciesMeta": { + "picomatch": { + "optional": true + } + } + }, + "node_modules/fsevents": { + "version": "2.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/fsevents/-/fsevents-2.3.3.tgz", + "integrity": "sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw==", + "hasInstallScript": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^8.16.0 || ^10.6.0 || >=11.0.0" + } + }, + "node_modules/graceful-fs": { + "version": "4.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/graceful-fs/-/graceful-fs-4.2.11.tgz", + "integrity": "sha512-RbJ5/jmFcNNCcDV5o9eTnBLJ/HszWV0P73bc+Ff4nS/rJj+YaS6IGyiOL0VoBYX+l1Wrl3k63h/KrH+nhJ0XvQ==", + "license": "ISC" + }, + "node_modules/jiti": { + "version": "2.7.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/jiti/-/jiti-2.7.0.tgz", + "integrity": "sha512-AC/7JofJvZGrrneWNaEnJeOLUx+JlGt7tNa0wZiRPT4MY1wmfKjt2+6O2p2uz2+skll8OZZmJMNqeke7kKbNgQ==", + "bin": { + "jiti": "lib/jiti-cli.mjs" + } + }, + "node_modules/lightningcss": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss/-/lightningcss-1.33.0.tgz", + "integrity": "sha512-WkUDrojuJs0xkgGf2udWxa3yGBRxPtxUkB79i6aCZLRgc7PM8fZe9TosfPDcvEpQZbuFASnHYmRLBLUbmLOIIA==", + "license": "MPL-2.0", + "dependencies": { + "detect-libc": "^2.0.3" + }, + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + }, + "optionalDependencies": { + "lightningcss-android-arm64": "1.33.0", + "lightningcss-darwin-arm64": "1.33.0", + "lightningcss-darwin-x64": "1.33.0", + "lightningcss-freebsd-x64": "1.33.0", + "lightningcss-linux-arm-gnueabihf": "1.33.0", + "lightningcss-linux-arm64-gnu": "1.33.0", + "lightningcss-linux-arm64-musl": "1.33.0", + "lightningcss-linux-x64-gnu": "1.33.0", + "lightningcss-linux-x64-musl": "1.33.0", + "lightningcss-win32-arm64-msvc": "1.33.0", + "lightningcss-win32-x64-msvc": "1.33.0" + } + }, + "node_modules/lightningcss-android-arm64": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-android-arm64/-/lightningcss-android-arm64-1.33.0.tgz", + "integrity": "sha512-gEpRTalKdosp4Bb8qWtc2iOgE5SeIHlpS1up9bFq2wAyYhl1UdTObYiHe98zEM9SQvSoqQZ1IQD0JNpg3Ml5pg==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-darwin-arm64": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-darwin-arm64/-/lightningcss-darwin-arm64-1.33.0.tgz", + "integrity": "sha512-Sciaz8eenNTKn9b3t7+xr0ipTp9YxKQY4npwQ3mrRuL0BAVHBLyZxofhaKBAVtzmtRZ/zTyo0/to4B1uWG/Djg==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-darwin-x64": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-darwin-x64/-/lightningcss-darwin-x64-1.33.0.tgz", + "integrity": "sha512-Z5UPAxzrjlWNNyGy6i65cJzzvgJ5D3T6wMvs+gWpY9d7qRhANrxqAp6LhxIgZhWEw18RfJTGcRxjuLIBr+m8XQ==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-freebsd-x64": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-freebsd-x64/-/lightningcss-freebsd-x64-1.33.0.tgz", + "integrity": "sha512-QQM/Ti/hQajJwCY+RiWuCZ9sdtI/XQk7nDK5vC8kkdwixezOlDgvDx7+RT+QjK6FcFT4MpsuoBnHIo/O3StRRg==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-linux-arm-gnueabihf": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-arm-gnueabihf/-/lightningcss-linux-arm-gnueabihf-1.33.0.tgz", + "integrity": "sha512-N7FVBe6iS24MlM6R/4RBTxGhQheZGs7tiQ9U32UtF75NzP5Q7xWPRqLBCKxlRQRk3rY1jCIPLzx7WzOhuUIRLQ==", + "cpu": [ + "arm" + ], + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-linux-arm64-gnu": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-arm64-gnu/-/lightningcss-linux-arm64-gnu-1.33.0.tgz", + "integrity": "sha512-j2v/itmy4HlNxlc6voKXYgBqNi0Ng2LShg4z7GufpEgs05P+2suBVyi9I6YHq5uoVFx9ETin3eCEhLVyXGQnKg==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-linux-arm64-musl": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-arm64-musl/-/lightningcss-linux-arm64-musl-1.33.0.tgz", + "integrity": "sha512-yiO5ROMuYQgXbC60yjZU5CYSFZGKXL0HFATXt9mHJn1+zW55oCtMI9NfcVhYLMFDL7gV7oBPon/EmMMGg2OvtQ==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-linux-x64-gnu": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-x64-gnu/-/lightningcss-linux-x64-gnu-1.33.0.tgz", + "integrity": "sha512-ar+Ju7LmcN0Jo4FpL4hpFybwNG9/3A/Br5KW2n2jyODg3MEZXaDYADdemoNS+BDNfMgKvylJLj4S5tyRActuAg==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-linux-x64-musl": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-linux-x64-musl/-/lightningcss-linux-x64-musl-1.33.0.tgz", + "integrity": "sha512-RYiYbkokw0trfKqqzfF55lginwEPrD3OJDfTuJzFs1MK6iFnDenaz1fqLLtX4ITG3OktJQXOeTaw1awrBAlZPw==", + "cpu": [ + "x64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-win32-arm64-msvc": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-win32-arm64-msvc/-/lightningcss-win32-arm64-msvc-1.33.0.tgz", + "integrity": "sha512-1K+MPfLSFVpphzpdbfkhlWk6wBrTObBzS2T6db10PNOZgR9GoVsAWzwNyuhUYYbTp23j+4RrncfujZ4uAzXvwA==", + "cpu": [ + "arm64" + ], + "license": "MPL-2.0", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/lightningcss-win32-x64-msvc": { + "version": "1.33.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/lightningcss-win32-x64-msvc/-/lightningcss-win32-x64-msvc-1.33.0.tgz", + "integrity": "sha512-OlEICDx/Xl0FqSp4bry8zFnCvGpig3Gl4gCquvYwHuqJKEC1+n9NgDniFvqHGmMv1ZkqDJrDqKKSykTDX+ehuA==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 12.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/parcel" + } + }, + "node_modules/magic-string": { + "version": "0.30.21", + "resolved": "https://mirrors.cloud.tencent.com/npm/magic-string/-/magic-string-0.30.21.tgz", + "integrity": "sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ==", + "license": "MIT", + "dependencies": { + "@jridgewell/sourcemap-codec": "^1.5.5" + } + }, + "node_modules/muggle-string": { + "version": "0.4.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/muggle-string/-/muggle-string-0.4.1.tgz", + "integrity": "sha512-VNTrAak/KhO2i8dqqnqnAHOa3cYBwXEZe9h+D5h/1ZqFSTEFHdM65lR7RoIqq3tBBYavsOXV84NoHXZ0AkPyqQ==", + "dev": true + }, + "node_modules/nanoid": { + "version": "3.3.19", + "resolved": "https://mirrors.cloud.tencent.com/npm/nanoid/-/nanoid-3.3.19.tgz", + "integrity": "sha512-Y2tUNy4ouw6tq5oDSKeQYGOyhkUBhNOcGV/02KC+6kd9eDGqdZd++mjMiIDilrBYvjEnCYvVtsuHCuP+okSfug==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/ai" + } + ], + "bin": { + "nanoid": "bin/nanoid.cjs" + }, + "engines": { + "node": "^10 || ^12 || ^13.7 || ^14 || >=15.0.1" + } + }, + "node_modules/path-browserify": { + "version": "1.0.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/path-browserify/-/path-browserify-1.0.1.tgz", + "integrity": "sha512-b7uo2UCUOYZcnF/3ID0lulOJi/bafxa1xPe7ZPsammBSpjSWQkjNxlt635YGS2MiR9GjvuXCtz2emr3jbsz98g==", + "dev": true, + "license": "MIT" + }, + "node_modules/picocolors": { + "version": "1.1.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/picocolors/-/picocolors-1.1.1.tgz", + "integrity": "sha512-xceH2snhtb5M9liqDsmEw56le376mTZkEX/jEb/RxNFyegNul7eNslCXP9FDj/Lcu0X8KEyMceP2ntpaHrDEVA==", + "license": "ISC" + }, + "node_modules/picomatch": { + "version": "4.0.7", + "resolved": "https://mirrors.cloud.tencent.com/npm/picomatch/-/picomatch-4.0.7.tgz", + "integrity": "sha512-qcJu88Q2IWqJsDD529JKMdwGm/dvInW4HvQnRwiH9JtihJvzGOscDtHE3x1pBKeUOTysQ8kVmLnJ2kJu7yhcGA==", + "engines": { + "node": ">=12" + }, + "funding": { + "url": "https://github.com/sponsors/jonschlinkert" + } + }, + "node_modules/postcss": { + "version": "8.5.28", + "resolved": "https://mirrors.cloud.tencent.com/npm/postcss/-/postcss-8.5.28.tgz", + "integrity": "sha512-RRuzqDtt5Y9h3quz5hWhK+TPnsmVs6WwSU6LkJMeY4HstUEDuYTG8UJSdawMRzmzAtV+KEoG8N3Qg2qLy5vM/A==", + "funding": [ + { + "type": "opencollective", + "url": "https://opencollective.com/postcss/" + }, + { + "type": "tidelift", + "url": "https://tidelift.com/funding/github/npm/postcss" + }, + { + "type": "github", + "url": "https://github.com/sponsors/ai" + } + ], + "dependencies": { + "nanoid": "^3.3.18", + "picocolors": "^1.1.1", + "source-map-js": "^1.2.1" + }, + "engines": { + "node": "^10 || ^12 || >=14" + } + }, + "node_modules/rolldown": { + "version": "1.2.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/rolldown/-/rolldown-1.2.11.tgz", + "integrity": "sha512-qpSwIyz0jHQq5qXBTNxFmE6664rJ7O+4TvPFOiOaBSrz8IOHc1koKKSqTM2H6u1UG1+TveuC6vaDHKXFOvb1Kw==", + "license": "MIT", + "dependencies": { + "@oxc-project/types": "=0.151.0", + "@rolldown/pluginutils": "^1.0.0" + }, + "bin": { + "rolldown": "bin/cli.mjs" + }, + "engines": { + "node": "^20.19.0 || >=22.12.0" + }, + "optionalDependencies": { + "@rolldown/binding-android-arm-eabi": "1.2.11", + "@rolldown/binding-android-arm64": "1.2.11", + "@rolldown/binding-darwin-arm64": "1.2.11", + "@rolldown/binding-darwin-x64": "1.2.11", + "@rolldown/binding-freebsd-x64": "1.2.11", + "@rolldown/binding-linux-arm-gnueabihf": "1.2.11", + "@rolldown/binding-linux-arm64-gnu": "1.2.11", + "@rolldown/binding-linux-arm64-musl": "1.2.11", + "@rolldown/binding-linux-ppc64-gnu": "1.2.11", + "@rolldown/binding-linux-s390x-gnu": "1.2.11", + "@rolldown/binding-linux-x64-gnu": "1.2.11", + "@rolldown/binding-linux-x64-musl": "1.2.11", + "@rolldown/binding-openharmony-arm64": "1.2.11", + "@rolldown/binding-win32-arm64-msvc": "1.2.11", + "@rolldown/binding-win32-x64-msvc": "1.2.11" + } + }, + "node_modules/source-map-js": { + "version": "1.2.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/source-map-js/-/source-map-js-1.2.1.tgz", + "integrity": "sha512-UXWMKhLOwVKb728IUtQPXxfYU+usdybtUrK/8uGE8CQMvrhOpwvzDBwj0QhSL7MQc7vIsISBG8VQ8+IDQxpfQA==", + "license": "BSD-3-Clause", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/tailwindcss": { + "version": "4.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/tailwindcss/-/tailwindcss-4.3.3.tgz", + "integrity": "sha512-gOhV3P7ufE62QDGg1zVaTgCR+EtPv92k2nIhVcVKcLmxT1sUBsQGhnZj175j+MqRt4zLF7ic+sCYjfhxMxj7YQ==" + }, + "node_modules/tapable": { + "version": "2.3.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/tapable/-/tapable-2.3.3.tgz", + "integrity": "sha512-uxc/zpqFg6x7C8vOE7lh6Lbda8eEL9zmVm/PLeTPBRhh1xCgdWaQ+J1CUieGpIfm2HdtsUpRv+HshiasBMcc6A==", + "license": "MIT", + "engines": { + "node": ">=6" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/webpack" + } + }, + "node_modules/tinyglobby": { + "version": "0.2.17", + "resolved": "https://mirrors.cloud.tencent.com/npm/tinyglobby/-/tinyglobby-0.2.17.tgz", + "integrity": "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g==", + "license": "MIT", + "dependencies": { + "fdir": "^6.5.0", + "picomatch": "^4.0.4" + }, + "engines": { + "node": ">=12.0.0" + }, + "funding": { + "url": "https://github.com/sponsors/SuperchupuDev" + } + }, + "node_modules/tslib": { + "version": "2.3.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/tslib/-/tslib-2.3.0.tgz", + "integrity": "sha512-N82ooyxVNm6h1riLCoyS9e3fuJ3AMG2zIZs2Gd1ATcSFjSA23Q0fzjjZeh0jbJvWVDZ0cJT8yaNNaaXHzueNjg==" + }, + "node_modules/typescript": { + "version": "6.0.3", + "resolved": "https://mirrors.cloud.tencent.com/npm/typescript/-/typescript-6.0.3.tgz", + "integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==", + "devOptional": true, + "license": "Apache-2.0", + "bin": { + "tsc": "bin/tsc", + "tsserver": "bin/tsserver" + }, + "engines": { + "node": ">=14.17" + } + }, + "node_modules/undici-types": { + "version": "7.24.6", + "resolved": "https://mirrors.cloud.tencent.com/npm/undici-types/-/undici-types-7.24.6.tgz", + "integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==", + "devOptional": true, + "license": "MIT" + }, + "node_modules/vite": { + "version": "8.3.1", + "resolved": "https://mirrors.cloud.tencent.com/npm/vite/-/vite-8.3.1.tgz", + "integrity": "sha512-/bvH9E9tmCXRGp2uXY3WbOldqpTwFkbha/8ANaEQ6VkxhH60KyqLwgZq6lG2y+4uT55x9+9eUHMpQ7uGnOCKjA==", + "license": "MIT", + "dependencies": { + "lightningcss": "^1.33.0", + "picomatch": "^4.0.7", + "postcss": "^8.5.28", + "rolldown": "~1.2.9", + "tinyglobby": "^0.2.17" + }, + "bin": { + "vite": "bin/vite.js" + }, + "engines": { + "node": "^20.19.0 || >=22.12.0" + }, + "funding": { + "url": "https://github.com/vitejs/vite?sponsor=1" + }, + "optionalDependencies": { + "fsevents": "~2.3.3" + }, + "peerDependencies": { + "@types/node": "^20.19.0 || >=22.12.0", + "@vitejs/devtools": "^0.7.1", + "esbuild": "^0.27.0 || ^0.28.0", + "jiti": ">=1.21.0", + "less": "^4.0.0", + "sass": "^1.70.0", + "sass-embedded": "^1.70.0", + "stylus": ">=0.54.8", + "sugarss": "^5.0.0", + "terser": "^5.16.0", + "tsx": "^4.8.1", + "yaml": "^2.4.2" + }, + "peerDependenciesMeta": { + "@types/node": { + "optional": true + }, + "@vitejs/devtools": { + "optional": true + }, + "esbuild": { + "optional": true + }, + "jiti": { + "optional": true + }, + "less": { + "optional": true + }, + "sass": { + "optional": true + }, + "sass-embedded": { + "optional": true + }, + "stylus": { + "optional": true + }, + "sugarss": { + "optional": true + }, + "terser": { + "optional": true + }, + "tsx": { + "optional": true + }, + "yaml": { + "optional": true + } + } + }, + "node_modules/vscode-uri": { + "version": "3.2.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/vscode-uri/-/vscode-uri-3.2.0.tgz", + "integrity": "sha512-m2gXo3bn0G1kT9InzMf07fTbqMbGtyckj3bH5ktLO+1Ssv+yiATZ4dhwaQv9UZWxJh6E9IFGnQyjgWVDWVBDrg==", + "dev": true, + "license": "MIT" + }, + "node_modules/vue": { + "version": "3.5.43", + "resolved": "https://mirrors.cloud.tencent.com/npm/vue/-/vue-3.5.43.tgz", + "integrity": "sha512-o5qZoksdnjIKvW1srZ3ab7pcDNYAerBjRe54D0LBLfRdCYFrSgBHVXokMas35czQc0//lmx4/tuY4ZNQ+Rf2Ng==", + "dependencies": { + "@vue/compiler-dom": "3.5.43", + "@vue/compiler-sfc": "3.5.43", + "@vue/runtime-dom": "3.5.43", + "@vue/server-renderer": "3.5.43", + "@vue/shared": "3.5.43" + }, + "peerDependencies": { + "typescript": "*" + }, + "peerDependenciesMeta": { + "typescript": { + "optional": true + } + } + }, + "node_modules/vue-tsc": { + "version": "3.3.11", + "resolved": "https://mirrors.cloud.tencent.com/npm/vue-tsc/-/vue-tsc-3.3.11.tgz", + "integrity": "sha512-gOb0B9rtU2+f1dszwPqSH5kAieIF9ReeLhD3kSRNHv5WZZUQz/JdVXW0RTdqhNTMlQkqKzrTTviqKr/4FYZraQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "@volar/typescript": "2.4.28", + "@vue/language-core": "3.3.11" + }, + "bin": { + "vue-tsc": "bin/vue-tsc.js" + }, + "peerDependencies": { + "typescript": ">=5.0.0" + } + }, + "node_modules/zrender": { + "version": "6.1.0", + "resolved": "https://mirrors.cloud.tencent.com/npm/zrender/-/zrender-6.1.0.tgz", + "integrity": "sha512-oEGMDB6pOP2S6OwRR4PdVv610zrjnA3Bh+JnSG12fYJlBKjtNAoEb5fSUoCOOINlH96I2fU38/A2UpRKs67xYQ==", + "license": "BSD-3-Clause", + "dependencies": { + "tslib": "2.3.0" + } + } + } +} diff --git a/frontend/package.json b/frontend/package.json new file mode 100644 index 0000000..695b096 --- /dev/null +++ b/frontend/package.json @@ -0,0 +1,25 @@ +{ + "name": "frontend", + "private": true, + "version": "0.0.0", + "type": "module", + "scripts": { + "dev": "vite", + "build": "vue-tsc -b && vite build", + "preview": "vite preview" + }, + "dependencies": { + "@tailwindcss/vite": "^4.3.3", + "echarts": "^6.1.0", + "tailwindcss": "^4.3.3", + "vue": "^3.5.42" + }, + "devDependencies": { + "@types/node": "^24.13.3", + "@vitejs/plugin-vue": "^6.0.8", + "@vue/tsconfig": "^0.9.1", + "typescript": "~6.0.2", + "vite": "^8.3.0", + "vue-tsc": "^3.3.11" + } +} diff --git a/frontend/public/favicon.svg b/frontend/public/favicon.svg new file mode 100644 index 0000000..6893eb1 --- /dev/null +++ b/frontend/public/favicon.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/frontend/public/icons.svg b/frontend/public/icons.svg new file mode 100644 index 0000000..e952219 --- /dev/null +++ b/frontend/public/icons.svg @@ -0,0 +1,24 @@ + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/frontend/src/App.vue b/frontend/src/App.vue new file mode 100644 index 0000000..5dbdadd --- /dev/null +++ b/frontend/src/App.vue @@ -0,0 +1,59 @@ + + + diff --git a/frontend/src/api.ts b/frontend/src/api.ts new file mode 100644 index 0000000..fd51cb9 --- /dev/null +++ b/frontend/src/api.ts @@ -0,0 +1,85 @@ +import type { + AssignmentItem, + HistoryResponse, + RuntimeSettings, + SettingsPatch, + StatusSnapshot, +} from './types' + +async function request(path: string, init?: RequestInit): Promise { + const resp = await fetch(path, { + headers: { 'Content-Type': 'application/json' }, + ...init, + }) + if (!resp.ok) { + // 把后端的 detail 原样带出来 —— 排障时这句话比状态码有用得多 + const text = await resp.text().catch(() => '') + throw new Error(text || `HTTP ${resp.status}`) + } + return (await resp.json()) as T +} + +export const api = { + status: () => request('/api/status'), + + setMode: (mode: 'auto' | 'manual') => + request<{ ok: boolean; mode: string }>('/api/mode', { + method: 'POST', + body: JSON.stringify({ mode }), + }), + + /** duty 传 null = 把该风扇位交回 BMC 自动控制 */ + setManualDuty: (slot: string, duty: number | null) => + request<{ ok: boolean; slot: string; duty: number | null }>('/api/manual', { + method: 'POST', + body: JSON.stringify({ slot, duty }), + }), + + /** 紧急回落:把所有风扇位交回 BMC。安全方向操作,任何时候都可调用 */ + restoreAuto: () => + request<{ ok: boolean; output: string; rc: number }>('/api/restore-auto', { + method: 'POST', + }), + + /** 历史趋势:后端代理 Prometheus query_range,一次拿回全部曲线 */ + history: (minutes = 30) => + request(`/api/history?minutes=${minutes}`), + + /** + * 更新「散热源 → 风扇位」的分配(GPU 为主体,它挑自己的风扇接口)。 + * + * ⚠️ **全量语义** —— 传的是完整列表,后端会整体覆盖,所以每次都要把 + * 所有源带上(只传一个会把其它的丢掉)。 + */ + setAssignments: (assignments: AssignmentItem[]) => + request<{ ok: boolean; assignments: AssignmentItem[] }>('/api/assignments', { + method: 'PUT', + body: JSON.stringify({ assignments }), + }), + + /** 操作审计(IPMI 写入 / API 调用 / 启停) */ + audit: (limit = 100) => + request<{ records: AuditRecord[] }>(`/api/audit?limit=${limit}`), + + /** 读取运行时设置(值来自 SQLite,界面上改过的即权威值) */ + getSettings: () => request('/api/settings'), + + /** + * 修改运行时设置(**部分更新**,只传要改的键)。 + * 改完立即生效,不需要重启服务。 + */ + patchSettings: (patch: SettingsPatch) => + request<{ ok: boolean; settings: RuntimeSettings }>('/api/settings', { + method: 'PATCH', + body: JSON.stringify(patch), + }), +} + +export interface AuditRecord { + id: number + ts: number + kind: string + actor: string + detail: string + ok: boolean +} diff --git a/frontend/src/assets/vite.svg b/frontend/src/assets/vite.svg new file mode 100644 index 0000000..5101b67 --- /dev/null +++ b/frontend/src/assets/vite.svg @@ -0,0 +1 @@ +Vite diff --git a/frontend/src/components/BindingWizard.vue b/frontend/src/components/BindingWizard.vue new file mode 100644 index 0000000..1c3d951 --- /dev/null +++ b/frontend/src/components/BindingWizard.vue @@ -0,0 +1,219 @@ + + + diff --git a/frontend/src/components/CurveChart.vue b/frontend/src/components/CurveChart.vue new file mode 100644 index 0000000..4cb2029 --- /dev/null +++ b/frontend/src/components/CurveChart.vue @@ -0,0 +1,110 @@ + + + diff --git a/frontend/src/components/FanPanel.vue b/frontend/src/components/FanPanel.vue new file mode 100644 index 0000000..db9ccfe --- /dev/null +++ b/frontend/src/components/FanPanel.vue @@ -0,0 +1,156 @@ + + + diff --git a/frontend/src/components/GpuCards.vue b/frontend/src/components/GpuCards.vue new file mode 100644 index 0000000..b5969c8 --- /dev/null +++ b/frontend/src/components/GpuCards.vue @@ -0,0 +1,109 @@ + + + diff --git a/frontend/src/components/SettingsPanel.vue b/frontend/src/components/SettingsPanel.vue new file mode 100644 index 0000000..9a265ff --- /dev/null +++ b/frontend/src/components/SettingsPanel.vue @@ -0,0 +1,388 @@ + + + diff --git a/frontend/src/components/Sidebar.vue b/frontend/src/components/Sidebar.vue new file mode 100644 index 0000000..6769f65 --- /dev/null +++ b/frontend/src/components/Sidebar.vue @@ -0,0 +1,137 @@ + + + diff --git a/frontend/src/components/TemperaturePanel.vue b/frontend/src/components/TemperaturePanel.vue new file mode 100644 index 0000000..f1fc448 --- /dev/null +++ b/frontend/src/components/TemperaturePanel.vue @@ -0,0 +1,98 @@ + + + diff --git a/frontend/src/components/TopBar.vue b/frontend/src/components/TopBar.vue new file mode 100644 index 0000000..6bcc20d --- /dev/null +++ b/frontend/src/components/TopBar.vue @@ -0,0 +1,140 @@ + + + diff --git a/frontend/src/components/TrendChart.vue b/frontend/src/components/TrendChart.vue new file mode 100644 index 0000000..b5542d8 --- /dev/null +++ b/frontend/src/components/TrendChart.vue @@ -0,0 +1,137 @@ + + + diff --git a/frontend/src/composables/useChart.ts b/frontend/src/composables/useChart.ts new file mode 100644 index 0000000..d729c7e --- /dev/null +++ b/frontend/src/composables/useChart.ts @@ -0,0 +1,46 @@ +import { onBeforeUnmount, onMounted, ref, type Ref } from 'vue' +import * as echarts from 'echarts/core' +import { LineChart } from 'echarts/charts' +import { + GridComponent, + LegendComponent, + MarkPointComponent, + TooltipComponent, +} from 'echarts/components' +import { CanvasRenderer } from 'echarts/renderers' + +// 按需注册 —— 只引这个面板真正用到的,别把整个 echarts 打进来 +echarts.use([ + LineChart, + GridComponent, + LegendComponent, + MarkPointComponent, + TooltipComponent, + CanvasRenderer, +]) + +export type EChartsInstance = echarts.ECharts + +/** 把一个 DOM 元素变成自适应尺寸的 ECharts 实例(卸载时自动销毁) */ +export function useChart(el: Ref) { + const chart = ref(null) + let observer: ResizeObserver | null = null + + onMounted(() => { + if (!el.value) return + chart.value = echarts.init(el.value) + + // 用 ResizeObserver 而不是 window.resize:面板在栅格里, + // 窗口没变但容器宽度也可能变(比如出现滚动条) + observer = new ResizeObserver(() => chart.value?.resize()) + observer.observe(el.value) + }) + + onBeforeUnmount(() => { + observer?.disconnect() + chart.value?.dispose() + chart.value = null + }) + + return chart +} diff --git a/frontend/src/composables/useRealtime.ts b/frontend/src/composables/useRealtime.ts new file mode 100644 index 0000000..fa0d824 --- /dev/null +++ b/frontend/src/composables/useRealtime.ts @@ -0,0 +1,119 @@ +import { onUnmounted, ref } from 'vue' +import { api } from '../api' +import type { StatusSnapshot } from '../types' + +/** + * 实时状态订阅。 + * + * **WebSocket 为主,轮询兜底**:WS 断了之后立刻降级成 5 秒一次轮询, + * 同时每 3 秒尝试重连 —— 界面不该因为 WS 抖动就停摆。 + * + * 注意后端推的是**完整快照**(2 秒一次),不是增量。快照很小(两张卡 + + * 四个风扇位),全量推送上位比维护增量状态简单得多,也不会有状态不一致。 + */ +export function useRealtime() { + const snapshot = ref(null) + const connected = ref(false) + const lastError = ref(null) + /** 上一次成功收到数据的时间戳,用来判断数据是否"不新鲜了" */ + const lastUpdate = ref(null) + + let ws: WebSocket | null = null + let reconnectTimer: number | null = null + let pollTimer: number | null = null + let disposed = false + + function wsUrl(): string { + const proto = location.protocol === 'https:' ? 'wss:' : 'ws:' + return `${proto}//${location.host}/api/ws` + } + + function connect(): void { + if (disposed) return + try { + ws = new WebSocket(wsUrl()) + } catch { + scheduleReconnect() + return + } + + ws.onopen = () => { + connected.value = true + lastError.value = null + stopPolling() + } + + ws.onmessage = (ev: MessageEvent) => { + try { + snapshot.value = JSON.parse(ev.data) as StatusSnapshot + lastUpdate.value = Date.now() + } catch { + /* 坏帧忽略,别把整个界面搞崩 */ + } + } + + ws.onclose = () => { + connected.value = false + scheduleReconnect() + } + + ws.onerror = () => { + lastError.value = 'WebSocket 连接异常' + // 具体重连交给随后的 onclose + } + } + + function scheduleReconnect(): void { + if (disposed || reconnectTimer !== null) return + startPolling() + reconnectTimer = window.setTimeout(() => { + reconnectTimer = null + connect() + }, 3000) + } + + function startPolling(): void { + if (pollTimer !== null) return + pollTimer = window.setInterval(() => { + void (async () => { + try { + snapshot.value = await api.status() + lastUpdate.value = Date.now() + lastError.value = null + } catch (e) { + lastError.value = e instanceof Error ? e.message : String(e) + } + })() + }, 5000) + } + + function stopPolling(): void { + if (pollTimer !== null) { + clearInterval(pollTimer) + pollTimer = null + } + } + + function close(): void { + disposed = true + stopPolling() + if (reconnectTimer !== null) clearTimeout(reconnectTimer) + ws?.close() + } + + /** 主动拉一次 —— 用户刚做完调控操作时调用,别干等下一个推送周期 */ + async function refresh(): Promise { + try { + snapshot.value = await api.status() + lastUpdate.value = Date.now() + lastError.value = null + } catch (e) { + lastError.value = e instanceof Error ? e.message : String(e) + } + } + + connect() + onUnmounted(close) + + return { snapshot, connected, lastError, lastUpdate, refresh } +} diff --git a/frontend/src/main.ts b/frontend/src/main.ts new file mode 100644 index 0000000..2425c0f --- /dev/null +++ b/frontend/src/main.ts @@ -0,0 +1,5 @@ +import { createApp } from 'vue' +import './style.css' +import App from './App.vue' + +createApp(App).mount('#app') diff --git a/frontend/src/pages/DashboardPage.vue b/frontend/src/pages/DashboardPage.vue new file mode 100644 index 0000000..bddf788 --- /dev/null +++ b/frontend/src/pages/DashboardPage.vue @@ -0,0 +1,19 @@ + + + diff --git a/frontend/src/pages/FansPage.vue b/frontend/src/pages/FansPage.vue new file mode 100644 index 0000000..69b92c7 --- /dev/null +++ b/frontend/src/pages/FansPage.vue @@ -0,0 +1,36 @@ + + + diff --git a/frontend/src/pages/SettingsPage.vue b/frontend/src/pages/SettingsPage.vue new file mode 100644 index 0000000..9ef2a42 --- /dev/null +++ b/frontend/src/pages/SettingsPage.vue @@ -0,0 +1,12 @@ + + + diff --git a/frontend/src/style.css b/frontend/src/style.css new file mode 100644 index 0000000..759ca0f --- /dev/null +++ b/frontend/src/style.css @@ -0,0 +1,67 @@ +@import "tailwindcss"; + +/* + * 设计系统 —— 现代 SaaS 风(Linear / Vercel / Stripe Dashboard 一路) + * + * 分层原则: + * 页面底色 zinc-50 (灰纸感,让白卡片浮出来) + * 卡片 white + zinc-200 细边框 + rounded-xl,不用重阴影 + * 文字层级 zinc-900 → zinc-500 → zinc-400(靠字重与灰度,不靠字号乱跳) + * 强调 zinc-900 主按钮;语义色只给状态(emerald 正常 / amber 警告 / + * red 危急 / blue 信息),别把面板搞花 + * 数字 .tnum 等宽字形 —— 读数跳动时不会左右晃 + */ + +html, +body, +#app { + height: 100%; +} + +body { + margin: 0; + font-family: + ui-sans-serif, system-ui, -apple-system, "Segoe UI", "PingFang SC", + "Hiragino Sans GB", "Microsoft YaHei", sans-serif; + -webkit-font-smoothing: antialiased; + text-rendering: optimizeLegibility; + background-color: var(--color-zinc-50); + color: var(--color-zinc-900); +} + +/* 数字用等宽字形,读数跳动时不会左右晃 */ +.tnum { + font-variant-numeric: tabular-nums; +} + +/* 统一焦点环:键盘导航时才显示,鼠标点击不闪 */ +:focus-visible { + outline: 2px solid var(--color-zinc-900); + outline-offset: 1px; + border-radius: 4px; +} + +/* 细滚动条 —— 监控面板常驻屏幕,粗滚动条很破坏质感 */ +* { + scrollbar-width: thin; + scrollbar-color: var(--color-zinc-300) transparent; +} +*::-webkit-scrollbar { + width: 8px; + height: 8px; +} +*::-webkit-scrollbar-thumb { + background: var(--color-zinc-300); + border-radius: 8px; +} +*::-webkit-scrollbar-thumb:hover { + background: var(--color-zinc-400); +} +*::-webkit-scrollbar-track { + background: transparent; +} + +::selection { + background: var(--color-zinc-900); + color: white; +} diff --git a/frontend/src/theme.ts b/frontend/src/theme.ts new file mode 100644 index 0000000..fb9fc34 --- /dev/null +++ b/frontend/src/theme.ts @@ -0,0 +1,52 @@ +/** + * ECharts 主题 token —— 与 Tailwind 的 zinc/语义色同源。 + * 两个图表(趋势 / 曲线)共用,保证整页图表一个味儿。 + */ + +export const CHART = { + /** 文字:zinc-500 的近亲 */ + text: '#71717a', + /** 弱文字:zinc-400 */ + textFaint: '#a1a1aa', + /** 网格线:zinc-100 */ + grid: '#f4f4f5', + /** 分割轴:zinc-200 */ + axis: '#e4e4e7', + /** tooltip 背景 */ + tooltipBg: '#ffffff', + tooltipBorder: '#e4e4e7', + + /** + * 系列色板 —— 温度暖色、转速冷色的语义保留: + * GPU 温度橙/红、CPU 温度琥珀、风扇转速蓝/青,够区分且不刺眼。 + */ + series: ['#f97316', '#dc2626', '#f59e0b', '#0ea5e9', '#6366f1', '#14b8a6', '#a855f7'], +} as const + +/** 两张图共用的公共片段(网格 / 坐标轴 / tooltip 基调) */ +export const chartBase = { + animation: false, + textStyle: { fontFamily: 'inherit' }, + grid: { left: 52, right: 56, top: 36, bottom: 28 }, + legend: { + top: 0, + left: 0, + itemWidth: 12, + itemHeight: 2, + itemGap: 14, + icon: 'rect', + textStyle: { fontSize: 11, color: CHART.text }, + }, + tooltip: { + trigger: 'axis' as const, + backgroundColor: CHART.tooltipBg, + borderColor: CHART.tooltipBorder, + borderWidth: 1, + padding: [8, 12], + textStyle: { fontSize: 11, color: '#18181b' }, + extraCssText: 'box-shadow: 0 4px 12px rgba(0,0,0,.08); border-radius: 8px;', + axisPointer: { type: 'line' as const, lineStyle: { color: CHART.axis } }, + }, + axisLabel: { fontSize: 11, color: CHART.textFaint }, + splitLine: { lineStyle: { color: CHART.grid } }, +} diff --git a/frontend/src/types.ts b/frontend/src/types.ts new file mode 100644 index 0000000..e50bb02 --- /dev/null +++ b/frontend/src/types.ts @@ -0,0 +1,177 @@ +/** + * 与后端 app/controller.py 的 describe() 输出一一对应。 + * 改后端字段时记得同步这里。 + */ + +export interface GpuInfo { + uuid: string + /** 短 uuid,如 e49ed30f —— 界面上用这个,完整 uuid 太长 */ + short_uuid: string + index: number + pci_bus_id: string + model: string + temperature: number | null + power_watts: number | null + utilization: number | null + memory_used_mib: number | null + memory_total_mib: number | null + memory_percent: number | null +} + +export interface FanInfo { + slot: string + /** 当前下发的占空比;null 表示该位交回 BMC 自动 */ + duty: number | null + temperature: number | null + curve_index: number | null + updated_ts: number | null + rpm: number | null + /** 温度源类型:gpu / gpu_group / cpu */ + source: string + /** 占用这个风扇位的源的 key(空 = 没被分配,程序不接管) */ + owner_key: string + /** 绑定的 GPU UUID(一对一绑定时有值) */ + bound_uuids: string[] + /** 人话说明:这一路温度是从哪儿来的(后端算好后给) */ + bound_detail: string +} + +/** + * 一个散热源的风扇分配 —— **主体是源(GPU),它去挑风扇接口**。 + * kind: gpu = 某张具体的卡;cpu = CPU 核温度 + */ +export interface AssignmentItem { + key: string + kind: 'gpu' | 'cpu' + gpu_uuid: string | null + /** 展示用的标签(后端拼好) */ + label: string + slots: string[] + temperature: number | null + online: boolean + /** 是否纳入管控(设置页可勾选;false = 分配保留但不驱动) */ + managed: boolean +} + +/** 可选的绑定目标 */ +export interface BindingOption { + uuid: string + short_uuid: string + label: string +} + +/** 可分配的风扇位(全部列出,没有转速读数的也在,rpm=null) */ +export interface FanSlotOption { + slot: string + rpm: number | null +} + +export interface CurvePoint { + temp: number + duty: number +} + +export interface CurveInfo { + points: CurvePoint[] + hysteresis: number + min_duty: number + max_duty: number +} + +/** CPU 核心温度(node_exporter 的 hwmon):Tctl / Tccd* */ +export interface CPUCoreTemp { + label: string + celsius: number | null +} + +/** BMC 板载温度传感器:MB Temp / CPU Temp / Card Side Temp / DDR4_* */ +export interface BoardTemp { + name: string + celsius: number | null + state: string +} + +export interface TemperatureInfo { + cpu_cores: CPUCoreTemp[] + board: BoardTemp[] + sources: { cpu: string; board: string } +} + +/** 运行时设置(可在界面上改,落 SQLite;配置文件只提供初始值) */ +export interface RuntimeSettings { + 'control.enabled': boolean + 'control.interval': number + /** 管控的 GPU UUID 列表;空数组 = 全部管控(保守默认) */ + 'control.managed_gpus': string[] + curve: CurveInfo + 'safety.emergency_temp': number + 'safety.emergency_resume_temp': number +} + +/** 部分更新用的补丁(只传要改的键) */ +export interface SettingsPatch { + control_enabled?: boolean + control_interval?: number + managed_gpus?: string[] + curve?: CurveInfo + emergency_temp?: number + emergency_resume_temp?: number +} + +export interface StatusSnapshot { + running: boolean + mode: 'auto' | 'manual' + interval: number + emergency: boolean + last_tick_ts: number | null + last_error: string | null + consecutive_failures: number + sources: { gpu: string; fan: string } + gpus: GpuInfo[] + fans: FanInfo[] + curve: CurveInfo + ipmi_target: Record + /** CPU 核温度 + BMC 板载温度。⚠️ GPU 温度在 gpus[] 里,不在这 */ + temperatures: TemperatureInfo + /** 「散热源 → 风扇位」的分配(主体是源,界面上按 GPU 一行一行选风扇) */ + assignments: AssignmentItem[] + /** 可分配的风扇位(全部列出,没读数的也在;rpm 供展示) */ + fan_slots: FanSlotOption[] + /** + * false = 首次使用,分配还没配过。 + * 此时后端**不会调档**,前端必须弹分配向导让用户自己指定。 + */ + bindings_configured: boolean + /** 可供绑定的目标(GPU 列表 + CPU 选项) */ + binding_options: { + gpus: BindingOption[] + cpu_label: string + } + /** 当前生效的运行时设置 */ + settings: RuntimeSettings +} + +/** 历史曲线中的一条 */ +export interface HistorySeriesItem { + key: string + label: string + unit: string + /** 挂在哪个 Y 轴上 —— 温度和转速量纲不同,必须分轴 */ + axis: 'left' | 'right' + /** [unix 毫秒, 值] */ + points: [number, number][] +} + +/** + * 历史数据响应。 + * + * 一次请求把图上所有曲线都取回来 —— 按指标分开请求会让前端发好几次 HTTP, + * 还得自己对齐时间轴,没有必要。 + * + * 数据由后端代理 Prometheus 的 `query_range` 得到,浏览器不直连 Prometheus + * (避免跨域,也别把地址暴露出去)。 + */ +export interface HistoryResponse { + minutes: number + series: HistorySeriesItem[] +} diff --git a/frontend/src/utils.ts b/frontend/src/utils.ts new file mode 100644 index 0000000..4cff238 --- /dev/null +++ b/frontend/src/utils.ts @@ -0,0 +1,48 @@ +/** 数值格式化与状态色阶 —— 多个组件共用 */ + +export function fmt( + value: number | null | undefined, + digits = 1, + suffix = '', +): string { + if (value === null || value === undefined || Number.isNaN(value)) return '—' + return value.toFixed(digits) + suffix +} + +export function fmtInt(value: number | null | undefined, suffix = ''): string { + if (value === null || value === undefined || Number.isNaN(value)) return '—' + return String(Math.round(value)) + suffix +} + +/** + * GPU 温度色阶。 + * Tesla T10 的 TjMax 约 89°C,且是**被动散热**(全靠机箱风扇吹), + * 所以阈值给得保守:70°C 起警告、80°C 起告警。 + */ +export function tempColor(t: number | null | undefined): string { + if (t === null || t === undefined) return 'text-slate-400' + if (t >= 80) return 'text-red-600' + if (t >= 70) return 'text-amber-600' + return 'text-emerald-600' +} + +export function tempBarColor(t: number | null | undefined): string { + if (t === null || t === undefined) return 'bg-slate-300' + if (t >= 80) return 'bg-red-500' + if (t >= 70) return 'bg-amber-500' + return 'bg-emerald-500' +} + +/** 距某个 unix 秒时间戳过去了多久(后端时间戳单位是秒) */ +export function secondsSince(ts: number | null | undefined): number | null { + if (!ts) return null + return Math.round(Date.now() / 1000 - ts) +} + +export function fmtAgo(ts: number | null | undefined): string { + const s = secondsSince(ts) + if (s === null) return '—' + if (s < 60) return `${s} 秒前` + if (s < 3600) return `${Math.floor(s / 60)} 分钟前` + return `${Math.floor(s / 3600)} 小时前` +} diff --git a/frontend/tsconfig.app.json b/frontend/tsconfig.app.json new file mode 100644 index 0000000..d72aa75 --- /dev/null +++ b/frontend/tsconfig.app.json @@ -0,0 +1,15 @@ +{ + "extends": "@vue/tsconfig/tsconfig.dom.json", + "compilerOptions": { + "tsBuildInfoFile": "./node_modules/.tmp/tsconfig.app.tsbuildinfo", + "types": ["vite/client"], + "allowArbitraryExtensions": true, + + /* Linting */ + "noUnusedLocals": true, + "noUnusedParameters": true, + "erasableSyntaxOnly": true, + "noFallthroughCasesInSwitch": true + }, + "include": ["src/**/*.ts", "src/**/*.tsx", "src/**/*.vue"] +} diff --git a/frontend/tsconfig.json b/frontend/tsconfig.json new file mode 100644 index 0000000..1ffef60 --- /dev/null +++ b/frontend/tsconfig.json @@ -0,0 +1,7 @@ +{ + "files": [], + "references": [ + { "path": "./tsconfig.app.json" }, + { "path": "./tsconfig.node.json" } + ] +} diff --git a/frontend/tsconfig.node.json b/frontend/tsconfig.node.json new file mode 100644 index 0000000..8455dcb --- /dev/null +++ b/frontend/tsconfig.node.json @@ -0,0 +1,23 @@ +{ + "compilerOptions": { + "tsBuildInfoFile": "./node_modules/.tmp/tsconfig.node.tsbuildinfo", + "target": "es2023", + "lib": ["ES2023"], + "types": ["node"], + "skipLibCheck": true, + + /* Bundler mode */ + "module": "nodenext", + "allowImportingTsExtensions": true, + "verbatimModuleSyntax": true, + "moduleDetection": "force", + "noEmit": true, + + /* Linting */ + "noUnusedLocals": true, + "noUnusedParameters": true, + "erasableSyntaxOnly": true, + "noFallthroughCasesInSwitch": true + }, + "include": ["vite.config.ts"] +} diff --git a/frontend/vite.config.ts b/frontend/vite.config.ts new file mode 100644 index 0000000..1c82799 --- /dev/null +++ b/frontend/vite.config.ts @@ -0,0 +1,33 @@ +import { fileURLToPath, URL } from 'node:url' +import { defineConfig } from 'vite' +import vue from '@vitejs/plugin-vue' +import tailwindcss from '@tailwindcss/vite' + +export default defineConfig({ + plugins: [vue(), tailwindcss()], + + resolve: { + alias: { + '@': fileURLToPath(new URL('./src', import.meta.url)), + }, + }, + + server: { + port: 5173, + // 开发时把 API 和 WebSocket 都代理到本机后端(uvicorn 跑在 :8765) + // 这样前端代码里一律写相对路径 /api/...,开发与生产完全一致 + proxy: { + '/api': { + target: 'http://127.0.0.1:8765', + changeOrigin: true, + ws: true, + }, + }, + }, + + build: { + // 产物直接落到后端静态目录 —— 「前后端一体」最终就是这一个服务 + outDir: '../app/static', + emptyOutDir: true, + }, +}) From 583b6e9b62b4843ce3ee3add61fadc50d2d9ed55 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E8=87=A3=E8=B6=85?= <517024110@qq.com> Date: Mon, 28 Sep 2026 20:26:37 +0800 Subject: [PATCH 06/13] =?UTF-8?q?=E4=BF=AE=E5=A4=8D:=20=E6=8E=A7=E5=88=B6?= =?UTF-8?q?=E6=A8=A1=E5=BC=8F=E6=8C=81=E4=B9=85=E5=8C=96=E5=88=B0=20SQLite?= =?UTF-8?q?=EF=BC=8C=E8=BF=90=E8=A1=8C=E6=97=B6=E7=8A=B6=E6=80=81=E4=BB=A5?= =?UTF-8?q?=E5=BA=93=E4=B8=BA=E5=94=AF=E4=B8=80=E6=9D=83=E5=A8=81?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - set_mode 切换后落库(此前只在内存,重启静默变回 auto, 控制器突然按曲线调档导致风扇行为突变) - 启动恢复: 先应用库里的设置(含 control.mode),再把最终生效值 全量回写 —— config.yaml 降级为首次初始化的种子, 不存在「回退到配置文件」的暗路径 - 设置页新增「控制模式」选择;模式切换写入审计日志 - 附带: 手动占空比入/出库时的曲线档位重置 --- app/api.py | 8 +++++++ app/controller.py | 20 ++++++++++++++++- app/main.py | 18 +++++++++++++--- frontend/src/components/SettingsPanel.vue | 26 +++++++++++++++++++++-- frontend/src/types.ts | 3 +++ 5 files changed, 69 insertions(+), 6 deletions(-) diff --git a/app/api.py b/app/api.py index dab2e12..36e2f82 100644 --- a/app/api.py +++ b/app/api.py @@ -165,6 +165,9 @@ class SettingsPatch(BaseModel): control_interval: float | None = Field( default=None, gt=0, le=3600, description="控制周期(秒)" ) + control_mode: Literal["auto", "manual"] | None = Field( + default=None, description="控制模式:auto=按曲线调档;manual=滑块直控" + ) curve: dict[str, Any] | None = Field( default=None, description="整条曲线 {points:[{temp,duty}], hysteresis, min_duty, max_duty}", @@ -211,6 +214,8 @@ async def patch_settings( patch["control.enabled"] = payload.control_enabled if payload.control_interval is not None: patch["control.interval"] = payload.control_interval + if payload.control_mode is not None: + patch["control.mode"] = payload.control_mode if payload.curve is not None: patch["curve"] = payload.curve if payload.emergency_temp is not None: @@ -359,10 +364,13 @@ async def history( @router.post("/mode", summary="切换控制模式") async def set_mode(payload: ModeRequest, request: Request) -> dict[str, Any]: controller = _controller(request) + store = getattr(request.app.state, "store", None) try: controller.set_mode(payload.mode) except ValueError as exc: raise HTTPException(status_code=400, detail=str(exc)) from exc + if store is not None: + store.log("api_call", "api", f"切换控制模式 → {payload.mode}") return {"ok": True, "mode": controller.mode} diff --git a/app/controller.py b/app/controller.py index f255b0d..4130dc1 100644 --- a/app/controller.py +++ b/app/controller.py @@ -221,10 +221,21 @@ def snapshot(self) -> ControllerSnapshot: # ------------------------------------------------------------ 模式切换 def set_mode(self, mode: str) -> None: - """切换 ``auto`` / ``manual``。 + """切换 ``auto`` / ``manual`` 并**持久化到 SQLite**。 切回 ``auto`` 时重置曲线状态,避免拿着旧的档位索引做滞回判断。 """ + self._switch_mode(mode) + # 模式是运行时状态,必须扛得住重启 —— 否则重启后静默变回 auto, + # 控制器突然开始按曲线调档,风扇行为突变(超哥 20:19 指出) + if self._store is not None: + try: + self._store.save_settings({"control.mode": mode}) + except Exception: # noqa: BLE001 - 持久化失败不该影响切换本身 + logger.exception("模式已切换,但持久化失败(重启后会回到旧值)") + + def _switch_mode(self, mode: str) -> None: + """纯内存的模式切换(:meth:`apply_settings` 启动恢复时复用,不落库)。""" if mode not in ("auto", "manual"): raise ValueError(f"不支持的模式: {mode!r}(可选 auto / manual)") if mode == self._mode: @@ -388,6 +399,12 @@ def apply_settings(self, settings: dict[str, Any]) -> None: Raises: ValueError: 值非法(API 层会转成 400)。 """ + if "control.mode" in settings: + value = settings["control.mode"] + if value not in ("auto", "manual"): + raise ValueError("control.mode 只能是 auto / manual") + self._switch_mode(value) + if "control.enabled" in settings: self._enabled = bool(settings["control.enabled"]) logger.info("控制总开关 → %s", "开启" if self._enabled else "关闭") @@ -454,6 +471,7 @@ def export_settings(self) -> dict[str, Any]: return { "control.enabled": self._enabled, "control.interval": self._interval, + "control.mode": self._mode, "control.managed_gpus": sorted(self._managed_gpus or []), "curve": self._curve.describe(), "safety.emergency_temp": self._emergency_temp, diff --git a/app/main.py b/app/main.py index ede3c9b..4161c1e 100644 --- a/app/main.py +++ b/app/main.py @@ -100,8 +100,11 @@ async def lifespan(app: FastAPI): store=store, ) - # 运行时设置:数据库里的值优先于配置文件(同样是单一数据源)。 - # ⚠️ **必须先于分配恢复** —— 管控范围(managed_gpus)是分配校验的前提, + # 运行时设置:**SQLite 是唯一权威**,config.yaml 只充当首次初始化的种子。 + # 启动时先应用库里已有的键,再把最终生效值**全量回写**—— + # 这样任何运行时状态(开关/模式/周期/曲线/阈值/管控范围)重启后都从库恢复, + # 不存在「回退到配置文件」的暗路径(超哥 20:19 明确要求)。 + # ⚠️ 这一段必须先于分配恢复 —— 管控范围(managed_gpus)是分配校验的前提, # 顺序反了的话,未管控卡的存量分配会把整份恢复作废(16:56 事故根因之一)。 stored_settings = store.load_settings() if stored_settings: @@ -111,7 +114,16 @@ async def lifespan(app: FastAPI): "已应用数据库里的运行时设置: %s", ", ".join(sorted(stored_settings)) ) except (ValueError, TypeError) as exc: - logger.error("数据库里的设置无效(%s)—— 沿用配置文件里的值", exc) + logger.error("数据库里的设置无效(%s)—— 缺失的键用配置文件种子补齐", exc) + try: + # 库里没有的键(首次启动 / 新增设置项)由配置文件种子补上并落库 + seed = controller.export_settings() + store.save_settings(seed) + newly = sorted(set(seed) - set(stored_settings or {})) + if newly: + logger.info("设置项首次落库(来自配置文件种子): %s", ", ".join(newly)) + except Exception: # noqa: BLE001 - 回写失败不影响启动 + logger.exception("设置回写 SQLite 失败") # 「散热源 → 风扇位」的分配以数据库为**唯一权威**(配置文件只声明管控哪些位)。 # diff --git a/frontend/src/components/SettingsPanel.vue b/frontend/src/components/SettingsPanel.vue index 9a265ff..02516d8 100644 --- a/frontend/src/components/SettingsPanel.vue +++ b/frontend/src/components/SettingsPanel.vue @@ -16,6 +16,7 @@ const error = ref(null) /** 本地草稿:整份设置的可编辑副本 */ const draft = ref<{ enabled: boolean + mode: 'auto' | 'manual' interval: number managedGpus: string[] points: CurvePoint[] @@ -31,6 +32,7 @@ function syncFromProps(s: RuntimeSettings | null) { const stored = s['control.managed_gpus'] ?? [] draft.value = { enabled: s['control.enabled'], + mode: s['control.mode'] ?? 'auto', interval: s['control.interval'], // 空列表语义 = 「全部管控」—— UI 上显示为全部勾选,保存时再还原成 [] managedGpus: stored.length ? [...stored] : props.gpus.map((g) => g.uuid), @@ -98,6 +100,7 @@ async function save() { props.gpus.every((g) => draft.value!.managedGpus.includes(g.uuid)) const patch: SettingsPatch = { control_enabled: draft.value.enabled, + control_mode: draft.value.mode, control_interval: draft.value.interval, managed_gpus: allSelected ? [] : draft.value.managedGpus, curve: { @@ -199,8 +202,8 @@ function reset() {

- -
+ +
+ +
- +
散热源 - - {{ fan.bound_detail || fan.owner_key }} - + + + + + - 未分配 —— 由 BMC 自动 + BMC 自动档(程序未接管 · 读不到 GPU 温度) + {{ fan.temperature.toFixed(1) }}°C @@ -149,7 +167,15 @@ function dutyLabel(fan: FanInfo): string { -

+

+ 三种「自动」别搞混: + 自动调档 + = 程序按 GPU 温度曲线算占空比,会随温度升降; + 手动定速 + = 占空比锁在你拖滑块定的值,温度再高也不提速; + BMC 自动档 + = 程序不接管,BMC 按主板自己的温度表转(它读不到 GPU)。 +
「哪张卡用哪个风扇」在上方分配面板里设置(以 GPU 为主体挑风扇接口)

diff --git a/frontend/src/components/SettingsPanel.vue b/frontend/src/components/SettingsPanel.vue index 9664f35..872171e 100644 --- a/frontend/src/components/SettingsPanel.vue +++ b/frontend/src/components/SettingsPanel.vue @@ -217,7 +217,9 @@ function reset() { 控制总开关 - {{ draft.enabled ? '开启 —— 按曲线自动调档' : '关闭 —— 完全不碰风扇' }} + + {{ draft.enabled ? '程序接管风扇(怎么定速看下面的模式)' : '程序完全不碰,全部交回 BMC 自动档' }} 控制模式 - {{ draft.mode === 'auto' ? '按曲线自动调档' : '滑块直控,不自动调档' }} + {{ + draft.mode === 'auto' + ? '自动调档:按温度曲线算占空比' + : '手动定速:锁定你设的占空比,温度涨也不提速' + }} diff --git a/frontend/src/components/TopBar.vue b/frontend/src/components/TopBar.vue index 6bcc20d..827c027 100644 --- a/frontend/src/components/TopBar.vue +++ b/frontend/src/components/TopBar.vue @@ -101,12 +101,12 @@ async function emergencyRestore() { :disabled="switching || !snapshot" :title=" mode === 'auto' - ? '当前按曲线自动调档,点击改为手动' - : '当前手动固定占空比,点击改回自动' + ? '当前:程序按 GPU 温度曲线算占空比(会随温度升降)。点击改为手动定速' + : '当前:占空比锁定在手动设定的值,GPU 温度再高也不提速。点击改回自动调档' " @click="toggleMode" > - 模式 · {{ mode === 'auto' ? '自动' : mode === 'manual' ? '手动' : '—' }} + {{ mode === 'auto' ? '自动调档' : mode === 'manual' ? '手动定速' : '—' }} From 5322ff9c88f49fe5387c93f657885128e1a8224e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E8=87=A3=E8=B6=85?= <517024110@qq.com> Date: Mon, 28 Sep 2026 21:26:46 +0800 Subject: [PATCH 09/13] =?UTF-8?q?=E6=96=B0=E5=A2=9E:=20=E6=9B=B2=E7=BA=BF?= =?UTF-8?q?=E8=AF=95=E7=AE=97=E9=A2=84=E8=A7=88=EF=BC=88/api/curve/preview?= =?UTF-8?q?=20=E8=B5=B0=E7=9C=9F=E5=AE=9E=E7=AE=97=E6=B3=95=EF=BC=89+=20Cu?= =?UTF-8?q?rveEditor=20=E7=BB=84=E4=BB=B6=E5=8C=96=EF=BC=8C=E8=B6=8B?= =?UTF-8?q?=E5=8A=BF=E5=9B=BE=E6=8B=86=E5=88=86=E4=B8=BA=E6=B8=A9=E5=BA=A6?= =?UTF-8?q?/=E8=BD=AC=E9=80=9F=E7=8B=AC=E7=AB=8B=E5=8F=8C=E5=9B=BE?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- app/api.py | 48 ++ app/controller.py | 60 +- app/curve.py | 72 ++- app/tests/test_core.py | 177 +++++- frontend/src/api.ts | 23 + frontend/src/components/CurveEditor.vue | 643 ++++++++++++++++++++++ frontend/src/components/SettingsPanel.vue | 117 +--- frontend/src/components/TrendChart.vue | 146 ++--- frontend/src/components/TrendPanel.vue | 123 +++++ frontend/src/composables/useChart.ts | 4 + frontend/src/pages/DashboardPage.vue | 4 +- frontend/src/theme.ts | 5 + frontend/src/types.ts | 15 + 13 files changed, 1199 insertions(+), 238 deletions(-) create mode 100644 frontend/src/components/CurveEditor.vue create mode 100644 frontend/src/components/TrendPanel.vue diff --git a/app/api.py b/app/api.py index 36e2f82..5ced706 100644 --- a/app/api.py +++ b/app/api.py @@ -187,6 +187,54 @@ class SettingsPatch(BaseModel): ) +class CurvePreviewPayload(BaseModel): + """曲线试算请求(**不落库** —— 纯粹是「改着看」)。 + + ``from`` 是 Python 关键字,字段用 ``from_`` + alias 收。 + """ + + model_config = {"populate_by_name": True} + + points: list[dict[str, Any]] = Field( + default_factory=list, description="折点 [{temp,duty}]" + ) + hysteresis: float = Field(default=3.0, description="滞回带 °C") + min_duty: int = Field(default=20, description="占空比下限 %") + max_duty: int = Field(default=100, description="占空比上限 %") + from_: float = Field(default=30.0, alias="from", description="采样起点 °C") + to: float = Field(default=100.0, description="采样终点 °C") + step: float = Field(default=1.0, description="采样步长 °C") + + +@router.post("/curve/preview", summary="试算一条曲线的输出(不落库)") +async def curve_preview( + payload: CurvePreviewPayload, request: Request +) -> dict[str, Any]: + """用**真实的控制器算法**算一遍还没保存的曲线,给前端画预览图。 + + 之所以不让前端照着公式自己画:阶梯语义(温度要达到折点才升档)+ 上下限 + 钳制这套逻辑只有一份实现才不会漂移。后端算出来的就是保存后真正会下发的。 + + 顺带承担校验职责 —— 折点温度重复 / 占空比越界 / 上下限倒置都会在这里 + 先撞成 400,界面上就能红出来,不用等点保存才报错。 + """ + controller = _controller(request) + try: + return controller.preview_curve( + { + "points": payload.points, + "hysteresis": payload.hysteresis, + "min_duty": payload.min_duty, + "max_duty": payload.max_duty, + "from": payload.from_, + "to": payload.to, + "step": payload.step, + } + ) + except (ValueError, TypeError) as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + + @router.get("/settings", summary="读取运行时设置") async def get_settings(request: Request) -> dict[str, Any]: """当前生效的设置。 diff --git a/app/controller.py b/app/controller.py index 5e6d9d1..40af668 100644 --- a/app/controller.py +++ b/app/controller.py @@ -21,7 +21,7 @@ from typing import Any from .config import AppConfig -from .curve import CurveState, FanCurve, build_curve_from_config +from .curve import MAX_TEMP, CurveState, FanCurve, build_curve_from_config from .ipmi import FAN_SLOT_INDEX, GPU_COOLING_SLOTS, IPMIClient from .safety import SafetyGuard from .sensors import ( @@ -434,10 +434,27 @@ def apply_settings(self, settings: dict[str, Any]) -> None: state.reset() logger.info("控制曲线已更新(%d 个折点)", len(points)) - if "safety.emergency_temp" in settings: - self._emergency_temp = float(settings["safety.emergency_temp"]) - if "safety.emergency_resume_temp" in settings: - self._emergency_resume = float(settings["safety.emergency_resume_temp"]) + # ⚠️ 两个阈值必须**一起**校验:解除温度 ≥ 触发温度的话,会陷入 + # 「一进紧急立刻解除 → 温度又上来 → 再进」的抖动脉冲(见 + # _update_emergency)。先算出新值、校验通过再落,避免改了一半被拒。 + if "safety.emergency_temp" in settings or "safety.emergency_resume_temp" in settings: + new_trigger = ( + float(settings["safety.emergency_temp"]) + if "safety.emergency_temp" in settings + else self._emergency_temp + ) + new_resume = ( + float(settings["safety.emergency_resume_temp"]) + if "safety.emergency_resume_temp" in settings + else self._emergency_resume + ) + if not 0 < new_resume < new_trigger <= MAX_TEMP: + raise ValueError( + f"紧急阈值需满足 0 < 解除({new_resume:g}°C)" + f" < 触发({new_trigger:g}°C)≤ {MAX_TEMP:g}°C" + ) + self._emergency_temp = new_trigger + self._emergency_resume = new_resume if "control.managed_gpus" in settings: value = settings["control.managed_gpus"] @@ -469,6 +486,39 @@ def apply_settings(self, settings: dict[str, Any]) -> None: a.slots = [] self._reindex() + def preview_curve(self, payload: dict[str, Any]) -> dict[str, Any]: + """试算一条**尚未保存**的曲线,返回采样点给前端画预览图。 + + 为什么不塞进 ``apply_settings``:预览是「改着看」的,每敲一个数字就 + 来一次;真落盘必须等用户点保存。这条路径只读不写, + ``self._curve`` 和所有 ``CurveState`` 都不受影响。 + + 复用 ``build_curve_from_config`` + ``FanCurve.sample`` 的完整链路 —— + 校验规则和真实曲线完全同一套,界面上预览到的就是保存后会跑的。 + """ + raw = payload or {} + points = raw.get("points") or [] + if not points: + raise ValueError("曲线至少要有一个折点") + + curve = build_curve_from_config( + points, + hysteresis=float(raw.get("hysteresis", 3.0)), + min_duty=int(raw.get("min_duty", 20)), + max_duty=int(raw.get("max_duty", 100)), + ) + return { + "points": [{"temp": p.temp, "duty": p.duty} for p in curve.points], + "hysteresis": curve.hysteresis, + "min_duty": curve.min_duty, + "max_duty": curve.max_duty, + "samples": curve.sample( + float(raw.get("from", 30.0)), + float(raw.get("to", 100.0)), + float(raw.get("step", 1.0)), + ), + } + def export_settings(self) -> dict[str, Any]: """导出当前运行时设置(供持久化到 SQLite / 给前端展示)。""" return { diff --git a/app/curve.py b/app/curve.py index 52edd74..816b140 100644 --- a/app/curve.py +++ b/app/curve.py @@ -28,6 +28,17 @@ logger = logging.getLogger(__name__) +#: 折点温度的合理上限。GPU 超过 110°C 基本已经触发硬件保护, +#: 再往上填没有意义,多半是手滑多打了一个 0。 +MAX_TEMP = 125.0 + +#: 滞回带上限。带太宽(比如 30°C)等于「降温永不降档」,风扇会一直顶着高档转。 +MAX_HYSTERESIS = 20.0 + +#: 预览采样的点数上限(防前端传极小步长把响应撑爆) +MAX_SAMPLES = 500 + + @dataclass(frozen=True) class CurvePoint: """曲线上的一个折点:温度达到 ``temp`` 时用 ``duty``。""" @@ -36,10 +47,14 @@ class CurvePoint: duty: int def __post_init__(self) -> None: + if not 0 <= self.temp <= MAX_TEMP: + raise ValueError(f"折点温度需在 0~{MAX_TEMP:g}°C 之间,收到 {self.temp}°C") if not 1 <= self.duty <= 100: raise ValueError(f"占空比需在 1~100 之间,收到 {self.duty}") + + @dataclass class CurveState: """曲线的运行状态。 @@ -77,10 +92,25 @@ def __init__( ordered = sorted(points, key=lambda p: p.temp) for prev, curr in zip(ordered, ordered[1:]): if prev.temp == curr.temp: - raise ValueError(f"曲线折点温度重复: {curr.temp}°C") + raise ValueError(f"曲线折点温度重复: {curr.temp:g}°C") + + # 上下限必须自身合法且有序 —— 否则钳制表达式 max(min, min(max, duty)) + # 会退化成「永远给下限」,界面上怎么改折点都没反应 + if not 1 <= min_duty <= 100: + raise ValueError(f"占空比下限需在 1~100 之间,收到 {min_duty}") + if not 1 <= max_duty <= 100: + raise ValueError(f"占空比上限需在 1~100 之间,收到 {max_duty}") + if min_duty > max_duty: + raise ValueError( + f"占空比下限({min_duty}%)不能高于上限({max_duty}%)" + ) + if not 0 <= hysteresis <= MAX_HYSTERESIS: + raise ValueError( + f"滞回带需在 0~{MAX_HYSTERESIS:g}°C 之间,收到 {hysteresis}°C" + ) self._points: tuple[CurvePoint, ...] = tuple(ordered) - self.hysteresis = max(0.0, hysteresis) + self.hysteresis = hysteresis self.min_duty = min_duty self.max_duty = max_duty @@ -148,6 +178,27 @@ def duty_at(self, temp: float) -> int: duty = self._points[self._index_for(temp)].duty return max(self.min_duty, min(self.max_duty, duty)) + def sample( + self, t_from: float, t_to: float, step: float = 1.0 + ) -> list[dict[str, float]]: + """按温度区间采样出一串 ``{temp, duty}``,供前端画预览图。 + + **为什么由后端算而不是前端照着公式自己画**:阶梯语义 + 上下限钳制 + 这套逻辑只有一份实现才不会漂移。前端拿到的就是控制器真正会下发的值。 + """ + if step <= 0: + raise ValueError(f"采样步长必须为正数,收到 {step}") + if t_to < t_from: + raise ValueError(f"采样区间终点({t_to:g})不能小于起点({t_from:g})") + + out: list[dict[str, float]] = [] + temp = t_from + # 上限兜底:前端传个 step=0.01 会把响应撑爆 + while temp <= t_to + 1e-9 and len(out) < MAX_SAMPLES: + out.append({"temp": round(temp, 2), "duty": self.duty_at(temp)}) + temp += step + return out + def describe(self) -> dict: """序列化给前端展示。""" return { @@ -159,6 +210,19 @@ def describe(self) -> dict: def build_curve_from_config(raw_points: Iterable[dict], **kwargs) -> FanCurve: - """从配置里的一串 ``{"temp": .., "duty": ..}`` 构造曲线。""" - points = [CurvePoint(float(p["temp"]), int(p["duty"])) for p in raw_points] + """从配置里的一串 ``{"temp": .., "duty": ..}`` 构造曲线。 + + 逐个显式转成数字而不是直接 ``int(...)`` —— 后者会把 70.5 静默砍成 70, + 界面上看着填了 70.5 实际生效的是 70,排障时很难发现。 + """ + points: list[CurvePoint] = [] + for raw in raw_points: + try: + temp = float(raw["temp"]) + duty = float(raw["duty"]) + except (KeyError, TypeError, ValueError) as exc: + raise ValueError(f"曲线折点格式不对:{raw!r}") from exc + if not float(duty).is_integer(): + raise ValueError(f"占空比必须是整数百分比,收到 {duty:g}") + points.append(CurvePoint(temp, int(duty))) return FanCurve(points, **kwargs) diff --git a/app/tests/test_core.py b/app/tests/test_core.py index 87aff7a..07b75ba 100644 --- a/app/tests/test_core.py +++ b/app/tests/test_core.py @@ -20,7 +20,12 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[2])) -from app.curve import CurvePoint, CurveState, FanCurve # noqa: E402 +from app.curve import ( # noqa: E402 + CurvePoint, + CurveState, + FanCurve, + build_curve_from_config, +) from app.ipmi import ( # noqa: E402 FAN_SLOT_INDEX, PAYLOAD_LEN, @@ -219,6 +224,118 @@ def test_state_reset(self) -> None: state.reset() self.assertIsNone(state.index) + def test_below_first_point_uses_first_duty(self) -> None: + """语义锁定:低于最低折点时给的是**首档**,不是 min_duty。 + + 这条很容易被误读成「低温 → 下限」,界面上必须写清楚。 + """ + curve = FanCurve([CurvePoint(50, 70)], min_duty=30) + self.assertEqual(curve.duty_at(40), 70, "低温应给首档 70%,不是下限 30%") + + def test_min_duty_above_max_is_rejected(self) -> None: + """上下限倒置会让钳制退化成常量(永远给下限),必须挡住。""" + with self.assertRaises(ValueError): + FanCurve([CurvePoint(50, 60)], min_duty=80, max_duty=40) + + def test_duty_must_be_integer_percent(self) -> None: + """70.5 不能静默砍成 70 —— 要么报错,要么按 70.5 处理,不能装没看见。""" + with self.assertRaises(ValueError): + build_curve_from_config([{"temp": 50, "duty": 70.5}]) + + def test_temp_out_of_range_is_rejected(self) -> None: + with self.assertRaises(ValueError): + CurvePoint(500, 50) + + def test_hysteresis_has_upper_bound(self) -> None: + """滞回带 50°C 等于「降温永不降档」,风扇会一直顶着高档转。""" + with self.assertRaises(ValueError): + FanCurve([CurvePoint(50, 60)], hysteresis=50) + + def test_sample_follows_step_curve(self) -> None: + curve = FanCurve([CurvePoint(50, 40), CurvePoint(80, 85)], min_duty=30) + samples = {round(s["temp"]): s["duty"] for s in curve.sample(40, 80, 10)} + self.assertEqual(samples[40], 40) # 低于首折点 + self.assertEqual(samples[50], 40) # 达到 50 → 首档 + self.assertEqual(samples[70], 40) # 还没到 80 + self.assertEqual(samples[80], 85) # 达到 80 → 升档 + + def test_sample_rejects_bad_range(self) -> None: + curve = FanCurve([CurvePoint(50, 40)]) + with self.assertRaises(ValueError): + curve.sample(80, 40, 1) + with self.assertRaises(ValueError): + curve.sample(40, 80, 0) + + +class TestCurvePreview(unittest.TestCase): + """「改着看」的曲线试算(preview_curve)—— 只读,不能污染运行状态。""" + + def _controller(self): + return _make_controller() + + def test_preview_matches_what_runtime_would_emit(self) -> None: + """预览出来的值 = 保存后真正会下发的值(同一套构造 + 钳制)。""" + c = self._controller() + payload = { + "points": [{"temp": 50, "duty": 40}, {"temp": 80, "duty": 85}], + "hysteresis": 3.0, + "min_duty": 30, + "max_duty": 100, + "from": 40, + "to": 80, + "step": 10, + } + preview = { + round(s["temp"]): s["duty"] for s in c.preview_curve(payload)["samples"] + } + + # 同一条曲线真正落进去,逐点比对 + c.apply_settings({"curve": payload}) + for temp, duty in preview.items(): + self.assertEqual( + c._curve.duty_at(float(temp)), duty, f"{temp}°C 预览与实跑不一致" + ) + + def test_preview_does_not_touch_runtime_curve(self) -> None: + """预览是「改着看」—— 没点保存就不能改到正在跑的曲线。""" + c = self._controller() + before = c._curve.describe() + c.preview_curve( + {"points": [{"temp": 30, "duty": 10}], "from": 30, "to": 60, "step": 10} + ) + self.assertEqual(c._curve.describe(), before) + + def test_preview_rejects_bad_draft(self) -> None: + """草稿非法就在试算阶段撞出来,别等点保存才 400。""" + c = self._controller() + for bad in ( + {"points": []}, # 空曲线 + {"points": [{"temp": 50, "duty": 40}, {"temp": 50, "duty": 60}]}, # 温度重复 + {"points": [{"temp": 50, "duty": 0}]}, # 占空比越界 + {"points": [{"temp": 50, "duty": 40}], "min_duty": 90, "max_duty": 50}, + ): + with self.assertRaises(ValueError): + c.preview_curve(bad) + + def test_emergency_resume_must_be_below_trigger(self) -> None: + """解除 ≥ 触发会让紧急状态一进就出,风扇 100% ↔ 曲线档反复横跳。""" + c = self._controller() + before = (c._emergency_temp, c._emergency_resume) + with self.assertRaises(ValueError): + c.apply_settings( + {"safety.emergency_temp": 70.0, "safety.emergency_resume_temp": 80.0} + ) + # 校验失败不能留下改了一半的状态 + self.assertEqual((c._emergency_temp, c._emergency_resume), before) + + def test_emergency_thresholds_accept_valid_pair(self) -> None: + c = self._controller() + c.apply_settings( + {"safety.emergency_temp": 88.0, "safety.emergency_resume_temp": 78.0} + ) + self.assertEqual(c._emergency_temp, 88.0) + self.assertEqual(c._emergency_resume, 78.0) + class TestPrometheusParsing(unittest.TestCase): """Prometheus 文本解析 —— 用 pve02 上抓到的真实格式。""" @@ -385,6 +502,37 @@ def test_rpm_values_are_correct(self) -> None: self.assertEqual(readings["REAR_FAN1"].rpm, 400.0) +def _make_controller(): + """构造一个最小可用的控制器(mock 掉 IPMI / 传感器)。 + + 抽成模块级函数 —— 曲线试算那组测试也要用,不复制一遍。 + """ + from app.config import AppConfig, FanBinding + from app.controller import FanController + from app.curve import build_curve_from_config + from app.ipmi import IPMIClient + from app.safety import SafetyGuard + from app.sensors import FanMetricsReader, GPUMetricsReader + + # 配置里显式预置两个位(模拟 config.yaml 占位;探测路径另有专项测试) + config = AppConfig(fans=[FanBinding(slot="FRNT_FAN1"), FanBinding(slot="REAR_FAN2")]) + ipmi = mock.MagicMock(spec=IPMIClient) + guard = mock.MagicMock(spec=SafetyGuard) + guard.engaged = False + gpu_reader = mock.MagicMock(spec=GPUMetricsReader) + fan_reader = mock.MagicMock(spec=FanMetricsReader) + # last_source 是实例属性(reader.read() 时才赋值),spec 的 Mock 上没有 + gpu_reader.last_source = "test" + fan_reader.last_source = "test" + curve = build_curve_from_config( + [ + {"temp": 45, "duty": 40}, + {"temp": 75, "duty": 80}, + ] + ) + return FanController(config, ipmi, guard, gpu_reader, fan_reader, curve) + + class TestSourceAssignments(unittest.TestCase): """「散热源 → 风扇位」分配模型(2026-09-28 倒置:GPU 是主体)。 @@ -397,32 +545,7 @@ class TestSourceAssignments(unittest.TestCase): GPU_B = "GPU-e49ed30f-f0f4-dc17-0225-2c1235602b39" def _controller(self): - from app.config import AppConfig, FanBinding - from app.controller import FanController - from app.curve import build_curve_from_config - from app.ipmi import IPMIClient - from app.safety import SafetyGuard - from app.sensors import FanMetricsReader, GPUMetricsReader - - # 配置里显式预置两个位(模拟 config.yaml 占位;探测路径另有专项测试) - config = AppConfig( - fans=[FanBinding(slot="FRNT_FAN1"), FanBinding(slot="REAR_FAN2")] - ) - ipmi = mock.MagicMock(spec=IPMIClient) - guard = mock.MagicMock(spec=SafetyGuard) - guard.engaged = False - gpu_reader = mock.MagicMock(spec=GPUMetricsReader) - fan_reader = mock.MagicMock(spec=FanMetricsReader) - # last_source 是实例属性(reader.read() 时才赋值),spec 的 Mock 上没有 - gpu_reader.last_source = "test" - fan_reader.last_source = "test" - curve = build_curve_from_config( - [ - {"temp": 45, "duty": 40}, - {"temp": 75, "duty": 80}, - ] - ) - return FanController(config, ipmi, guard, gpu_reader, fan_reader, curve) + return _make_controller() @staticmethod def _gpu(uuid: str, temp: float): diff --git a/frontend/src/api.ts b/frontend/src/api.ts index fd51cb9..3f97b02 100644 --- a/frontend/src/api.ts +++ b/frontend/src/api.ts @@ -1,5 +1,7 @@ import type { AssignmentItem, + CurvePoint, + CurvePreviewResponse, HistoryResponse, RuntimeSettings, SettingsPatch, @@ -57,6 +59,27 @@ export const api = { body: JSON.stringify({ assignments }), }), + /** + * 试算一条**还没保存**的曲线。 + * + * 走后端真实算法(阶梯语义 + 上下限钳制),保证「预览到的」就是「保存后会跑的」。 + * 顺带承担校验:折点温度重复 / 占空比越界 / 上下限倒置都会撞成 400, + * 错误信息就是 ``detail`` 原文,可以直接显示给用户。 + */ + previewCurve: (payload: { + points: CurvePoint[] + hysteresis: number + min_duty: number + max_duty: number + from: number + to: number + step: number + }) => + request('/api/curve/preview', { + method: 'POST', + body: JSON.stringify(payload), + }), + /** 操作审计(IPMI 写入 / API 调用 / 启停) */ audit: (limit = 100) => request<{ records: AuditRecord[] }>(`/api/audit?limit=${limit}`), diff --git a/frontend/src/components/CurveEditor.vue b/frontend/src/components/CurveEditor.vue new file mode 100644 index 0000000..4869de5 --- /dev/null +++ b/frontend/src/components/CurveEditor.vue @@ -0,0 +1,643 @@ + + + diff --git a/frontend/src/components/SettingsPanel.vue b/frontend/src/components/SettingsPanel.vue index 872171e..899b606 100644 --- a/frontend/src/components/SettingsPanel.vue +++ b/frontend/src/components/SettingsPanel.vue @@ -1,6 +1,7 @@ diff --git a/frontend/src/components/TrendPanel.vue b/frontend/src/components/TrendPanel.vue new file mode 100644 index 0000000..80af1c5 --- /dev/null +++ b/frontend/src/components/TrendPanel.vue @@ -0,0 +1,123 @@ + + + diff --git a/frontend/src/composables/useChart.ts b/frontend/src/composables/useChart.ts index d729c7e..4174ced 100644 --- a/frontend/src/composables/useChart.ts +++ b/frontend/src/composables/useChart.ts @@ -4,6 +4,8 @@ import { LineChart } from 'echarts/charts' import { GridComponent, LegendComponent, + MarkAreaComponent, + MarkLineComponent, MarkPointComponent, TooltipComponent, } from 'echarts/components' @@ -15,6 +17,8 @@ echarts.use([ GridComponent, LegendComponent, MarkPointComponent, + MarkLineComponent, + MarkAreaComponent, TooltipComponent, CanvasRenderer, ]) diff --git a/frontend/src/pages/DashboardPage.vue b/frontend/src/pages/DashboardPage.vue index bddf788..1d66040 100644 --- a/frontend/src/pages/DashboardPage.vue +++ b/frontend/src/pages/DashboardPage.vue @@ -1,7 +1,7 @@