From b9c5a6236f7d660c27fab91faa306fe0b9249f59 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E8=87=A3=E8=B6=85?= <517024110@qq.com> Date: Tue, 29 Sep 2026 19:40:50 +0800 Subject: [PATCH] =?UTF-8?q?=E5=AE=89=E5=85=A8=E8=84=B1=E6=95=8F:=20?= =?UTF-8?q?=E6=B8=85=E9=99=A4=E5=85=AC=E5=BC=80=E4=BB=93=E5=BA=93=E4=B8=AD?= =?UTF-8?q?=E7=9A=84=E5=86=85=E7=BD=91=E6=8B=93=E6=89=91=E4=BF=A1=E6=81=AF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 真实内网 IP 全部替换为 RFC 5737 文档段(192.0.2.10/11/20),涉及文档、 代码注释、config.yaml、deploy.sh 默认值 - 主机名 pve02 按语境移除/泛化(文档、代码注释、CI 注释、前端 UI) - DEPLOY.md 删除 BMC 凭据示例,强调凭据不得入公开仓库 - app/config.yaml 转为模板形态(示例地址 + 替换提醒);deploy.sh 与 DEPLOY.md 升级流程排除 config.yaml,防止模板覆盖现场配置 - deploy.sh 的 CONN 改为必填环境变量,不再默认指向作者机器 - 前端侧边栏地址改为动态取当前访问 origin,不再硬编码 --- .github/workflows/ci.yml | 2 +- CLAUDE.md | 18 +++++------ DEPLOY.md | 47 ++++++++++++++--------------- README.md | 4 +-- app/__init__.py | 2 +- app/config.py | 6 ++-- app/config.yaml | 16 +++++----- app/ipmi.py | 8 ++--- app/main.py | 2 +- app/sensors.py | 2 +- app/tests/test_core.py | 6 ++-- deploy.sh | 13 +++++--- fanController/base_controller.py | 10 +++--- fanController/epycd8_controller.py | 14 ++++----- frontend/src/components/Sidebar.vue | 7 +++-- utils/prometheus_client.py | 6 ++-- 16 files changed, 84 insertions(+), 79 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9559442..ca78c65 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -23,7 +23,7 @@ jobs: - uses: actions/setup-python@v5 with: - python-version: "3.13" # 对齐 pve02 线上版本 + python-version: "3.13" # 对齐线上生产版本 - name: 后端单元测试 run: | diff --git a/CLAUDE.md b/CLAUDE.md index 1e5321d..f75bbae 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -166,22 +166,22 @@ cat logs/fancontroller.log.YYYY-MM-DD ```yaml prometheus: - base_url: "http://192.168.6.31:30091" + base_url: "http://192.0.2.20:30091" timeout: 10 servers: - type: epycd8 prometheus: # per-server 覆盖(多机场景必需) - fan_instance: "192.168.6.7:9290" # 风扇转速 ← ipmi_exporter - temp_instance: "192.168.6.7:9100" # CPU 温度 ← node_exporter - gpu_instance: "192.168.6.7:9400" # GPU 温度 ← DCGM exporter + fan_instance: "192.0.2.10:9290" # 风扇转速 ← ipmi_exporter + temp_instance: "192.0.2.10:9100" # CPU 温度 ← node_exporter + gpu_instance: "192.0.2.10:9400" # GPU 温度 ← DCGM exporter ``` -| 数据 | 指标 | instance(pve02 实测) | +| 数据 | 指标 | instance(实测示例) | |------|------|----------------------| -| 风扇转速 | `ipmi_fan_speed_rpm{name="FRNT_FAN1"}` | `192.168.6.7:9290` | -| CPU 温度 | `node_hwmon_temp_celsius`(按语义标签 `label="Tctl"` 过滤) | `192.168.6.7:9100` | -| GPU 温度 | `DCGM_FI_DEV_GPU_TEMP` | `192.168.6.7:9400` | +| 风扇转速 | `ipmi_fan_speed_rpm{name="FRNT_FAN1"}` | `192.0.2.10:9290` | +| CPU 温度 | `node_hwmon_temp_celsius`(按语义标签 `label="Tctl"` 过滤) | `192.0.2.10:9100` | +| GPU 温度 | `DCGM_FI_DEV_GPU_TEMP` | `192.0.2.10:9400` | ⚠️ **三个 instance 对应三个不同的 exporter,混用会直接查不到数据。** 配置项 分别为 `fan_instance` / `temp_instance` / `gpu_instance`(笼统的 `instance` 仍 @@ -219,7 +219,7 @@ servers: 本仓库真正在用的是 `app/`(FastAPI 控制台)+ `frontend/`(Vue3 前端): -- 控速依据是 **GPU 温度**(DCGM),目标机 pve02 的机箱风扇,写 `ipmitool raw 0x3a 0x01` +- 控速依据是 **GPU 温度**(DCGM),目标机的机箱风扇,写 `ipmitool raw 0x3a 0x01` - 运行时状态(模式 / 管控 GPU / 分配关系 / 审计)**全部落 SQLite**,配置文件只是首次运行的种子 **实时数据源:三个外部 exporter(自备组件,不随仓库提供,装法不限)** diff --git a/DEPLOY.md b/DEPLOY.md index e9e3284..3d25161 100644 --- a/DEPLOY.md +++ b/DEPLOY.md @@ -1,9 +1,9 @@ # 部署手册 —— GPU 风扇控制台 -> 目标机:**pve02**(ASRock Rack EPYCD8,192.168.6.7,BMC 2.20) +> 目标机:你的 GPU 宿主机(下文 `<目标机>` / `<连接名>` 按实际替换) > 部署目录:`/opt/gpu-fan-console` | 服务名:`gpu-fan-console.service` -> 访问地址:`http://192.168.6.7:8765` -> 最后核对:2026-09-28(当时线上 Python 3.13.5 / fastapi 0.141.1 / uvicorn 0.53.0) +> 访问地址:`http://<目标机>:8765` +> 参考环境:Python 3.13.5 / fastapi 0.141.1 / uvicorn 0.53.0(2026-09-28 核对) --- @@ -37,11 +37,11 @@ FastAPI 挂 `StaticFiles` 一起发出去。所以: | 代码目录 | `/opt/gpu-fan-console` | 工作目录,service 的 `WorkingDirectory` | | 虚拟环境 | `/opt/gpu-fan-console/.venv` | Python 3.13.5,已装 fastapi / uvicorn / PyYAML | | 数据库 | `/opt/gpu-fan-console/app/data/fan-console.db` | **SQLite 是权威数据源**,升级时绝不能被覆盖 | -| 运行时配置 | `/opt/gpu-fan-console/app/config.yaml` | 只在**首次运行**当初始值用;之后改了不算数 | +| 运行时配置 | `/opt/gpu-fan-console/app/config.yaml` | 只在**首次运行**当初始值用;之后改了不算数。仓库里的版本是**模板**(示例地址),升级时解压要排除它(见 4.1 步骤 4) | | 主服务 | `/etc/systemd/system/gpu-fan-console.service` | `enabled` + `active`,`Restart=always` | | 看门狗 | `/etc/systemd/system/fan-watchdog.{service,timer}` | `enabled`,每 2 分钟查一次心跳 | | 心跳文件 | `/run/gpu-fan-console/heartbeat` | 主进程每轮写;过期则由看门狗强推 `8×0x00` 回落 | -| 依赖的 exporter | ipmi_exporter `:9290`、node_exporter `:9100`、DCGM `:9400` | 都在 pve02 本机,都在跑 | +| 依赖的 exporter | ipmi_exporter `:9290`、node_exporter `:9100`、DCGM `:9400` | 自备组件,装法不限,见 2.1 | ### 2.1 前置依赖:三个 exporter(自备组件,装法不限) @@ -72,12 +72,6 @@ curl -s localhost:9290/metrics | grep ipmi_fan_speed_rpm | head -1 curl -s localhost:9100/metrics | grep Tctl | head -1 ``` -> **pve02 当前实况**(仅本机维护参考,不是规定):dcgm-exporter 跑 docker -> (`nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless`,`--restart unless-stopped`, -> `--gpus all`);ipmi_exporter v1.10.1 与 node_exporter 跑 systemd,二进制在 -> `/usr/local/bin/`,node_exporter 带 `--collector.hwmon --collector.cpufreq`, -> ipmi_exporter 以 root 运行。 - --- ## 3. 首次部署(换机器 / 重装时才需要) @@ -87,7 +81,7 @@ curl -s localhost:9100/metrics | grep Tctl | head -1 # 它们是控制台的全部数据源,没有这一步控速无依据 # ① 建目录、建 venv -ssh pve02 +ssh <目标机> mkdir -p /opt/gpu-fan-console cd /opt/gpu-fan-console python3 -m venv .venv @@ -125,14 +119,17 @@ tar czf /tmp/gfc.tgz \ -C . app # 3) 传上去 -agentsshcli upload pve02 /tmp/gfc.tgz /root/gfc.tgz # 这条路现在常挂,见「坑 ②」 +agentsshcli upload <连接名> /tmp/gfc.tgz /root/gfc.tgz # 这条路现在常挂,见「坑 ②」 # 4) 远端解压 + 清旧前端产物(⚠️ 见「坑 ③」) -agentsshcli exec pve02 "tar xzf /root/gfc.tgz -C /opt/gpu-fan-console" +# 4) 远端解压 + 清旧前端产物(⚠️ 见「坑 ③」) +# 已配置过的机器升级时排除 config.yaml —— 仓库里的版本是模板(示例地址), +# 别把现场配好的 prometheus_url 等覆盖回示例值。首次部署才需要带上它。 +agentsshcli exec <连接名> "tar xzf /root/gfc.tgz -C /opt/gpu-fan-console --exclude='app/config.yaml'" # 5) 重启 + 自检 -agentsshcli exec pve02 "systemctl restart gpu-fan-console && sleep 4 && systemctl is-active gpu-fan-console" -curl -s http://192.168.6.7:8765/api/status | head -c 300 +agentsshcli exec <连接名> "systemctl restart gpu-fan-console && sleep 4 && systemctl is-active gpu-fan-console" +curl -s http://192.0.2.10:8765/api/status | head -c 300 ``` `deploy.sh` 就是把上面这套串起来,并且**内置了下面三个坑的绕过方案**。 @@ -188,10 +185,10 @@ split -b 30000 all.b64 ch_ # ⚠️ 单次命令行上限 32767 字符 i=0 for f in ch_*; do [ $i -eq 0 ] && R='>' || R='>>' - agentsshcli exec pve02 "printf '%s' '$(cat $f)' $R /root/pkg.b64" + agentsshcli exec <连接名> "printf '%s' '$(cat $f)' $R /root/pkg.b64" i=$((i+1)) done -agentsshcli exec pve02 "base64 -d /root/pkg.b64 > /root/pkg.tgz" +agentsshcli exec <连接名> "base64 -d /root/pkg.b64 > /root/pkg.tgz" ``` > 只发前端时先 `gzip -9` 再 base64,体积能砍到 1/3,分片数从 25 降到 10。 @@ -214,9 +211,9 @@ find -type f ! -name '新文件1' ! -name '新文件2' -delete |---|---|---| | 服务活着 | `systemctl is-active gpu-fan-console` | `active` | | 看门狗在跑 | `systemctl list-timers fan-watchdog.timer` | 有下次触发时间 | -| 页面 200 | `curl -s -o /dev/null -w '%{http_code}' http://192.168.6.7:8765/` | `200` | -| 静态资源对得上 | `curl -s http://192.168.6.7:8765/ \| grep -o 'index-[A-Za-z0-9_-]*\.\(js\|css\)'` | 与本地 `ls app/static/assets` 一致 | -| 曲线在控速 | `curl -s http://192.168.6.7:8765/api/status` | 受控位 `duty` 有值、`temperature` 与 GPU 温度对得上 | +| 页面 200 | `curl -s -o /dev/null -w '%{http_code}' http://192.0.2.10:8765/` | `200` | +| 静态资源对得上 | `curl -s http://192.0.2.10:8765/ \| grep -o 'index-[A-Za-z0-9_-]*\.\(js\|css\)'` | 与本地 `ls app/static/assets` 一致 | +| 曲线在控速 | `curl -s http://192.0.2.10:8765/api/status` | 受控位 `duty` 有值、`temperature` 与 GPU 温度对得上 | | 数据没丢 | 打开设置页 | 模式 / 管控 GPU / 分配关系都是部署前的样子 | | 启动顺序对 | `journalctl -u gpu-fan-console -n 30` | 先「已应用数据库里的设置」再「已应用数据库里的分配」 | | 回落动作在 | 同上,搜「安全护栏已激活」 | 有这行才说明退出时会回退 auto | @@ -232,9 +229,9 @@ git checkout <上一个好提交> ./deploy.sh # 数据回滚(SQLite 有备份时) -agentsshcli exec pve02 "systemctl stop gpu-fan-console" -agentsshcli exec pve02 "cp /root/fan-console.db.bak /opt/gpu-fan-console/app/data/fan-console.db" -agentsshcli exec pve02 "systemctl start gpu-fan-console" +agentsshcli exec <连接名> "systemctl stop gpu-fan-console" +agentsshcli exec <连接名> "cp /root/fan-console.db.bak /opt/gpu-fan-console/app/data/fan-console.db" +agentsshcli exec <连接名> "systemctl start gpu-fan-console" ``` **最坏情况的保险**:不管服务死没死,都可以直接让 BMC 接管 —— @@ -242,7 +239,7 @@ agentsshcli exec pve02 "systemctl start gpu-fan-console" ```bash # 远端执行(把 8 个字节全填 0x00 = 全交回 BMC 自动档) ipmitool raw 0x3a 0x01 0x00 0x00 0x00 0x00 0x00 0x00 0x00 0x00 -# 宿主都起不来时:从 BMC 独立地址 192.168.6.8 走,凭据 admin/admin +# 宿主都起不来时:从 BMC 独立管理地址走(BMC 与宿主 OS 相互独立,凭据自行保管,切勿写进公开文档) ``` --- diff --git a/README.md b/README.md index 59e45d8..1135a10 100644 --- a/README.md +++ b/README.md @@ -134,7 +134,7 @@ python -m unittest discover -s app/tests -t . -v 1. **后端单测**:`python -m unittest`(Python 3.13,对齐线上) 2. **前端构建**:`npm ci && npm run build`(自带 vue-tsc 全量类型检查) 3. **部署包**:产出 `gpu-fan-console-app.tgz`(成员路径 `app/...`,排除 `app/data`), - 挂在 workflow 的 Artifacts 里——下载后传到 pve02 解压重启即可,本机无需装 Node + 挂在 workflow 的 Artifacts 里——下载后传到目标机解压重启即可,本机无需装 Node 推送 main 时额外构建**双平台独立可执行文件**(PyInstaller);打 `v*` tag 自动创建 GitHub Release 并附上全部产物: @@ -201,7 +201,7 @@ alert: # 邮件告警(可选,默认关闭) max_failed_attempts: 3 email: { ... } # SMTP 配置,支持多收件人、1 小时防轰炸 prometheus: - base_url: "http://192.168.6.31:30091" + base_url: "http://192.0.2.20:30091" servers: - type: dell730 ip: "192.168.71.90" diff --git a/app/__init__.py b/app/__init__.py index ac3fb04..4467afe 100644 --- a/app/__init__.py +++ b/app/__init__.py @@ -5,7 +5,7 @@ 2. **API 服务**:REST + WebSocket 给前端 3. **静态托管**:直接伺服前端构建产物 -没有跨机通信、没有独立 agent —— 它只管 pve02 自己这台机器的风扇。 +没有跨机通信、没有独立 agent —— 它只管部署它的这台机器的风扇。 """ __version__ = "0.1.0" diff --git a/app/config.py b/app/config.py index 2b86494..4aae86a 100644 --- a/app/config.py +++ b/app/config.py @@ -56,11 +56,11 @@ class SourcesConfig(BaseModel): # 实时读数**不走**这里,因为 Prometheus 是 30s 快照(见 CLAUDE.md)。 # 但历史曲线恰恰是 Prometheus 的主场,所以单独配一套。 prometheus_url: str | None = None - #: GPU 温度在 Prometheus 里的 instance 标签,如 ``192.168.6.7:9400`` + #: GPU 温度在 Prometheus 里的 instance 标签,如 ``192.0.2.10:9400`` prometheus_gpu_instance: str | None = None - #: CPU 温度(node_exporter 的 hwmon / k10temp)的 instance 标签,如 ``192.168.6.7:9100`` + #: CPU 温度(node_exporter 的 hwmon / k10temp)的 instance 标签,如 ``192.0.2.10:9100`` prometheus_node_instance: str | None = None - #: 风扇转速的 instance 标签,如 ``192.168.6.7:9290`` + #: 风扇转速的 instance 标签,如 ``192.0.2.10:9290`` prometheus_fan_instance: str | None = None diff --git a/app/config.yaml b/app/config.yaml index ca46859..0b1100a 100644 --- a/app/config.yaml +++ b/app/config.yaml @@ -29,10 +29,11 @@ sources: http_timeout: 5 # 远程 BMC 兜底(宿主系统起不来时排障用)。默认走本地 in-band,不配这项。 + # ⚠️ 换成你自己的 BMC 地址与凭据;凭据绝不要提交到公开仓库。 # ipmi_remote: - # host: "192.168.6.8" - # user: "admin" - # password: "admin" + # host: "192.0.2.11" # 示例地址(RFC 5737 文档段),按实际替换 + # user: "your-bmc-user" + # password: "your-bmc-password" # --- 历史趋势(可选,只给 /api/history 用)--- # @@ -40,10 +41,11 @@ sources: # 而上面两个 endpoint 是直连 exporter,读的才是当下值。 # 但历史曲线恰恰是 Prometheus 的主场,所以单独配一套。 # 不配的话界面上的趋势图会显示"暂无历史数据",其余功能不受影响。 - prometheus_url: "http://192.168.6.31:30091" - prometheus_gpu_instance: "192.168.6.7:9400" # GPU 温度 ← DCGM exporter - prometheus_node_instance: "192.168.6.7:9100" # CPU 温度 ← node_exporter - prometheus_fan_instance: "192.168.6.7:9290" # 风扇转速 ← ipmi_exporter + # ⚠️ 下面的地址全部是示例(RFC 5737 文档段),按你的实际环境替换。 + prometheus_url: "http://192.0.2.20:30091" # 示例:你的 Prometheus 地址 + prometheus_gpu_instance: "192.0.2.10:9400" # GPU 温度 ← DCGM exporter + prometheus_node_instance: "192.0.2.10:9100" # CPU 温度 ← node_exporter + prometheus_fan_instance: "192.0.2.10:9290" # 风扇转速 ← ipmi_exporter curve: # 滞回带(°C):降温方向必须跌出这个带宽才降档,防止温度在阈值附近 diff --git a/app/ipmi.py b/app/ipmi.py index eabbdcf..568d40c 100644 --- a/app/ipmi.py +++ b/app/ipmi.py @@ -23,8 +23,8 @@ b1 CPU1_FAN1 b2 --(保留) b3 REAR_FAN1 - b4 REAR_FAN2 ← pve02 用于 Tesla T10 散热 - b5 FRNT_FAN1 ← pve02 用于 Tesla T10 散热 + b4 REAR_FAN2 ← 宿主机用于 Tesla T10 散热 + b5 FRNT_FAN1 ← 宿主机用于 Tesla T10 散热 b6 FRNT_FAN2 (未接风扇) b7 FRNT_FAN3 (未接风扇) b8 FRNT_FAN4 (未接风扇) @@ -148,9 +148,9 @@ def encode_duty(duty: int | None) -> int: class IPMIClient: """ipmitool 封装。 - 默认走**本地 in-band**(``/dev/ipmi0``,需要 root),这是 pve02 上的推荐用法: + 默认走**本地 in-band**(``/dev/ipmi0``,需要 root),这是推荐用法: 链路最短、无网络依赖。也支持 ``lanplus`` 远程模式指向 BMC 独立地址 - (``192.168.6.8``),用于宿主系统起不来时的带外兜底。 + (``192.0.2.11``),用于宿主系统起不来时的带外兜底。 """ def __init__( diff --git a/app/main.py b/app/main.py index 6ad250d..27bdcb4 100644 --- a/app/main.py +++ b/app/main.py @@ -186,7 +186,7 @@ async def lifespan(app: FastAPI): app = FastAPI( title="GPU Fan Console", - description="pve02 GPU 温度联动风扇控制台(单机应用)", + description="GPU 温度联动风扇控制台(单机应用)", version=__version__, lifespan=lifespan, ) diff --git a/app/sensors.py b/app/sensors.py index 1c4fd30..16c0e26 100644 --- a/app/sensors.py +++ b/app/sensors.py @@ -13,7 +13,7 @@ - 所有网络/子进程调用**带超时**。上游项目的教训:一个不带超时的阻塞读 能把整个控制线程静默挂死。 -关于 pve02 这块板子(ASRock Rack EPYCD8,BMC 固件 2.20)的实测要点: +关于本机这块板子(ASRock Rack EPYCD8,BMC 固件 2.20)的实测要点: - ``ipmitool sdr type fan`` 输出**五列**:``名称 | 传感器ID | 状态 | 阈值 | 读数``。 读数在**最后一列**。最初按「第二列是读数」写会把传感器 ID ``62h`` 当成 RPM。 diff --git a/app/tests/test_core.py b/app/tests/test_core.py index c71e6de..0c7ef80 100644 --- a/app/tests/test_core.py +++ b/app/tests/test_core.py @@ -338,7 +338,7 @@ def test_emergency_thresholds_accept_valid_pair(self) -> None: class TestPrometheusParsing(unittest.TestCase): - """Prometheus 文本解析 —— 用 pve02 上抓到的真实格式。""" + """Prometheus 文本解析 —— 用 真实设备抓到的格式。""" SAMPLE = """\ # HELP DCGM_FI_DEV_GPU_TEMP GPU temperature (in C). @@ -428,7 +428,7 @@ def test_missing_clock_stays_none(self) -> None: class TestSdrFanParsing(unittest.TestCase): """``ipmitool sdr type fan`` 输出解析。 - 样例取自 2026-09-28 在 pve02 上的真实输出 —— 这组用例的由来就是一个 + 样例取自 2026-09-28 在真实设备上的输出 —— 这组用例的由来就是一个 真实 bug:最初以为「第二列是读数」,结果把传感器 ID ``62h`` 当成了 RPM。 """ @@ -499,7 +499,7 @@ def test_no_reading_temperature(self) -> None: class TestFanReaderIpmitoolFallback(unittest.TestCase): """ipmitool 兜底路径 —— 必须跳过未接的风扇位。 - 样例是 2026-09-28 在 pve02 上抓的真实输出:14 个风扇传感器位里只有 + 样例是 2026-09-28 在真实设备上抓的输出:14 个风扇传感器位里只有 4 个有读数,其余全是 ``No Reading``。不跳过的话界面上会凭空多出 10 个空风扇位。 """ diff --git a/deploy.sh b/deploy.sh index 20302fa..ec0f178 100644 --- a/deploy.sh +++ b/deploy.sh @@ -1,24 +1,24 @@ #!/usr/bin/env bash # -# 一键部署 GPU 风扇控制台到 pve02 +# 一键部署 GPU 风扇控制台到目标机 # # ./deploy.sh 全量(后端 + 前端)→ 覆盖 → 重启 → 自检 # ./deploy.sh --static-only 只更新前端(不改后端,不重启,静态文件即时生效) # ./deploy.sh --no-build 跳过前端构建(复用 app/static 里已有的产物) # ./deploy.sh --no-restart 传完不重启(全量模式下慎用) # -# 可用环境变量覆盖默认值: -# CONN=pve02 REMOTE_DIR=/opt/gpu-fan-console SERVICE=gpu-fan-console URL=http://192.168.6.7:8765 +# 可用环境变量(CONN 必填,其余有默认值): +# CONN= REMOTE_DIR=/opt/gpu-fan-console SERVICE=gpu-fan-console URL=http://<目标机>:8765 # # 为什么不用 agentsshcli upload:它现在报「创建远端续传元数据失败」,虽然数据传完了 # 但整体返回失败(--no-cache 直连模式也无效,已实测)。所以走分片 base64,见 DEPLOY.md 坑 ②。 # set -euo pipefail -CONN="${CONN:-pve02}" +CONN="${CONN:?用法: CONN= ./deploy.sh}" REMOTE_DIR="${REMOTE_DIR:-/opt/gpu-fan-console}" SERVICE="${SERVICE:-gpu-fan-console}" -URL="${URL:-http://192.168.6.7:8765}" +URL="${URL:-http://127.0.0.1:8765}" # 单次命令行上限 32767 字符(Windows),留出余量 CHUNK_SIZE="${CHUNK_SIZE:-30000}" @@ -71,8 +71,11 @@ if [ "$STATIC_ONLY" -eq 1 ]; then tar czf "$PKG" -C . app/static else # ⚠️ --exclude='app/data' 是在保命:那是 SQLite 权威数据源,被覆盖等于丢失全部配置 + # ⚠️ --exclude='app/config.yaml':仓库里的是模板(示例地址),覆盖会毁掉现场配好的 + # prometheus_url 等本机值。首次部署请手动传一次完整 config.yaml 再改。 tar czf "$PKG" \ --exclude='__pycache__' --exclude='*.pyc' --exclude='app/data' \ + --exclude='app/config.yaml' \ -C . app fi echo " 包大小:$(wc -c <"$PKG") bytes" diff --git a/fanController/base_controller.py b/fanController/base_controller.py index 5ad4d28..497fe5b 100644 --- a/fanController/base_controller.py +++ b/fanController/base_controller.py @@ -21,7 +21,7 @@ def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True, auto (bool): 是否自动模式,True为自动模式,False为手动模式。 alert_config (dict, optional): 告警配置字典。 prometheus_config (dict, optional): Prometheus 查询配置,形如 - ``{'base_url': 'http://192.168.6.31:30091', 'instance': '192.168.6.7:9290'}``。 + ``{'base_url': 'http://192.0.2.20:30091', 'instance': '192.0.2.10:9290'}``。 风扇转速统一从这里取(见 :meth:`get_fan_rotational_speed`)。 """ self.platform_system = platform.system() @@ -130,11 +130,11 @@ def _pick_instance(self, *keys): ⚠️ **不同数据源的 instance 是不同的**,因为它们是各自的 exporter: ============ ================== ========================== - 数据 指标 instance(pve02 实测) + 数据 指标 instance(实测示例) ============ ================== ========================== - 风扇转速 ipmi_fan_speed_rpm ``192.168.6.7:9290`` - CPU 温度 node_hwmon_temp_* ``192.168.6.7:9100`` - GPU 温度 DCGM_FI_DEV_* ``192.168.6.7:9400`` + 风扇转速 ipmi_fan_speed_rpm ``192.0.2.10:9290`` + CPU 温度 node_hwmon_temp_* ``192.0.2.10:9100`` + GPU 温度 DCGM_FI_DEV_* ``192.0.2.10:9400`` ============ ================== ========================== 混用一个 ``instance`` 会直接查不到数据 —— 这个坑 2026-09-28 实现时踩到。 diff --git a/fanController/epycd8_controller.py b/fanController/epycd8_controller.py index ad88be2..fec7042 100644 --- a/fanController/epycd8_controller.py +++ b/fanController/epycd8_controller.py @@ -27,11 +27,11 @@ b1 CPU1_FAN1 b2 --(保留) b3 REAR_FAN1 - b4 REAR_FAN2 ← pve02 用于 Tesla T10 散热 - b5 FRNT_FAN1 ← pve02 用于 Tesla T10 散热 - b6 FRNT_FAN2 (pve02 未接风扇) - b7 FRNT_FAN3 (pve02 未接风扇) - b8 FRNT_FAN4 (pve02 未接风扇) + b4 REAR_FAN2 ← 宿主机用于 Tesla T10 散热 + b5 FRNT_FAN1 ← 宿主机用于 Tesla T10 散热 + b6 FRNT_FAN2 (未接风扇) + b7 FRNT_FAN3 (未接风扇) + b8 FRNT_FAN4 (未接风扇) """ from .base_controller import IPMIFanController @@ -94,7 +94,7 @@ def _build_temperature_query(self): 字段名,在本机型上承载的其实是 GPU 温度。 数据源同样是 Prometheus —— DCGM exporter 已接入 - (job ``dcgm-exporter-pve02``),指标 ``DCGM_FI_DEV_GPU_TEMP``。 + (job 名自定义,如 ``dcgm-exporter``),指标 ``DCGM_FI_DEV_GPU_TEMP``。 Returns: str | None: PromQL;未配置 ``gpu_instance`` 时返回 ``None``。 @@ -103,7 +103,7 @@ def _build_temperature_query(self): if not instance: self.logger.error( f"服务器 {self.ip}: 未配置 prometheus.gpu_instance,无法定位 " - f"DCGM 数据(例: 192.168.6.7:9400)" + f"DCGM 数据(例: 192.0.2.10:9400)" ) return None return f'DCGM_FI_DEV_GPU_TEMP{{instance="{instance}"}}' diff --git a/frontend/src/components/Sidebar.vue b/frontend/src/components/Sidebar.vue index 6769f65..5aa9114 100644 --- a/frontend/src/components/Sidebar.vue +++ b/frontend/src/components/Sidebar.vue @@ -14,6 +14,9 @@ defineProps<{ const emit = defineEmits<{ (e: 'navigate', p: PageKey): void }>() +// 同源部署:直接显示浏览器当前访问的地址,不写死任何主机名/IP +const accessAddress = window.location.host || '同源部署' + const NAV: { key: PageKey; label: string; hint: string }[] = [ { key: 'dashboard', label: '概览', hint: 'GPU 状态 · 温度 · 趋势' }, { key: 'fans', label: '风扇控制', hint: '分配 · 调速 · 曲线' }, @@ -38,7 +41,7 @@ const NAV: { key: PageKey; label: string; hint: string }[] = [

GPU 风扇控制台

- +
@@ -123,7 +126,7 @@ const NAV: { key: PageKey; label: string; hint: string }[] = [ /> {{ connected ? '实时连接' : '轮询模式' }}

-

192.168.6.7:8765

+

{{ accessAddress }}