From b9c5a6236f7d660c27fab91faa306fe0b9249f59 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?=E6=9D=8E=E8=87=A3=E8=B6=85?= <517024110@qq.com>
Date: Tue, 29 Sep 2026 19:40:50 +0800
Subject: [PATCH] =?UTF-8?q?=E5=AE=89=E5=85=A8=E8=84=B1=E6=95=8F:=20?=
=?UTF-8?q?=E6=B8=85=E9=99=A4=E5=85=AC=E5=BC=80=E4=BB=93=E5=BA=93=E4=B8=AD?=
=?UTF-8?q?=E7=9A=84=E5=86=85=E7=BD=91=E6=8B=93=E6=89=91=E4=BF=A1=E6=81=AF?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
- 真实内网 IP 全部替换为 RFC 5737 文档段(192.0.2.10/11/20),涉及文档、
代码注释、config.yaml、deploy.sh 默认值
- 主机名 pve02 按语境移除/泛化(文档、代码注释、CI 注释、前端 UI)
- DEPLOY.md 删除 BMC 凭据示例,强调凭据不得入公开仓库
- app/config.yaml 转为模板形态(示例地址 + 替换提醒);deploy.sh 与
DEPLOY.md 升级流程排除 config.yaml,防止模板覆盖现场配置
- deploy.sh 的 CONN 改为必填环境变量,不再默认指向作者机器
- 前端侧边栏地址改为动态取当前访问 origin,不再硬编码
---
.github/workflows/ci.yml | 2 +-
CLAUDE.md | 18 +++++------
DEPLOY.md | 47 ++++++++++++++---------------
README.md | 4 +--
app/__init__.py | 2 +-
app/config.py | 6 ++--
app/config.yaml | 16 +++++-----
app/ipmi.py | 8 ++---
app/main.py | 2 +-
app/sensors.py | 2 +-
app/tests/test_core.py | 6 ++--
deploy.sh | 13 +++++---
fanController/base_controller.py | 10 +++---
fanController/epycd8_controller.py | 14 ++++-----
frontend/src/components/Sidebar.vue | 7 +++--
utils/prometheus_client.py | 6 ++--
16 files changed, 84 insertions(+), 79 deletions(-)
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 9559442..ca78c65 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -23,7 +23,7 @@ jobs:
- uses: actions/setup-python@v5
with:
- python-version: "3.13" # 对齐 pve02 线上版本
+ python-version: "3.13" # 对齐线上生产版本
- name: 后端单元测试
run: |
diff --git a/CLAUDE.md b/CLAUDE.md
index 1e5321d..f75bbae 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -166,22 +166,22 @@ cat logs/fancontroller.log.YYYY-MM-DD
```yaml
prometheus:
- base_url: "http://192.168.6.31:30091"
+ base_url: "http://192.0.2.20:30091"
timeout: 10
servers:
- type: epycd8
prometheus: # per-server 覆盖(多机场景必需)
- fan_instance: "192.168.6.7:9290" # 风扇转速 ← ipmi_exporter
- temp_instance: "192.168.6.7:9100" # CPU 温度 ← node_exporter
- gpu_instance: "192.168.6.7:9400" # GPU 温度 ← DCGM exporter
+ fan_instance: "192.0.2.10:9290" # 风扇转速 ← ipmi_exporter
+ temp_instance: "192.0.2.10:9100" # CPU 温度 ← node_exporter
+ gpu_instance: "192.0.2.10:9400" # GPU 温度 ← DCGM exporter
```
-| 数据 | 指标 | instance(pve02 实测) |
+| 数据 | 指标 | instance(实测示例) |
|------|------|----------------------|
-| 风扇转速 | `ipmi_fan_speed_rpm{name="FRNT_FAN1"}` | `192.168.6.7:9290` |
-| CPU 温度 | `node_hwmon_temp_celsius`(按语义标签 `label="Tctl"` 过滤) | `192.168.6.7:9100` |
-| GPU 温度 | `DCGM_FI_DEV_GPU_TEMP` | `192.168.6.7:9400` |
+| 风扇转速 | `ipmi_fan_speed_rpm{name="FRNT_FAN1"}` | `192.0.2.10:9290` |
+| CPU 温度 | `node_hwmon_temp_celsius`(按语义标签 `label="Tctl"` 过滤) | `192.0.2.10:9100` |
+| GPU 温度 | `DCGM_FI_DEV_GPU_TEMP` | `192.0.2.10:9400` |
⚠️ **三个 instance 对应三个不同的 exporter,混用会直接查不到数据。** 配置项
分别为 `fan_instance` / `temp_instance` / `gpu_instance`(笼统的 `instance` 仍
@@ -219,7 +219,7 @@ servers:
本仓库真正在用的是 `app/`(FastAPI 控制台)+ `frontend/`(Vue3 前端):
-- 控速依据是 **GPU 温度**(DCGM),目标机 pve02 的机箱风扇,写 `ipmitool raw 0x3a 0x01`
+- 控速依据是 **GPU 温度**(DCGM),目标机的机箱风扇,写 `ipmitool raw 0x3a 0x01`
- 运行时状态(模式 / 管控 GPU / 分配关系 / 审计)**全部落 SQLite**,配置文件只是首次运行的种子
**实时数据源:三个外部 exporter(自备组件,不随仓库提供,装法不限)**
diff --git a/DEPLOY.md b/DEPLOY.md
index e9e3284..3d25161 100644
--- a/DEPLOY.md
+++ b/DEPLOY.md
@@ -1,9 +1,9 @@
# 部署手册 —— GPU 风扇控制台
-> 目标机:**pve02**(ASRock Rack EPYCD8,192.168.6.7,BMC 2.20)
+> 目标机:你的 GPU 宿主机(下文 `<目标机>` / `<连接名>` 按实际替换)
> 部署目录:`/opt/gpu-fan-console` | 服务名:`gpu-fan-console.service`
-> 访问地址:`http://192.168.6.7:8765`
-> 最后核对:2026-09-28(当时线上 Python 3.13.5 / fastapi 0.141.1 / uvicorn 0.53.0)
+> 访问地址:`http://<目标机>:8765`
+> 参考环境:Python 3.13.5 / fastapi 0.141.1 / uvicorn 0.53.0(2026-09-28 核对)
---
@@ -37,11 +37,11 @@ FastAPI 挂 `StaticFiles` 一起发出去。所以:
| 代码目录 | `/opt/gpu-fan-console` | 工作目录,service 的 `WorkingDirectory` |
| 虚拟环境 | `/opt/gpu-fan-console/.venv` | Python 3.13.5,已装 fastapi / uvicorn / PyYAML |
| 数据库 | `/opt/gpu-fan-console/app/data/fan-console.db` | **SQLite 是权威数据源**,升级时绝不能被覆盖 |
-| 运行时配置 | `/opt/gpu-fan-console/app/config.yaml` | 只在**首次运行**当初始值用;之后改了不算数 |
+| 运行时配置 | `/opt/gpu-fan-console/app/config.yaml` | 只在**首次运行**当初始值用;之后改了不算数。仓库里的版本是**模板**(示例地址),升级时解压要排除它(见 4.1 步骤 4) |
| 主服务 | `/etc/systemd/system/gpu-fan-console.service` | `enabled` + `active`,`Restart=always` |
| 看门狗 | `/etc/systemd/system/fan-watchdog.{service,timer}` | `enabled`,每 2 分钟查一次心跳 |
| 心跳文件 | `/run/gpu-fan-console/heartbeat` | 主进程每轮写;过期则由看门狗强推 `8×0x00` 回落 |
-| 依赖的 exporter | ipmi_exporter `:9290`、node_exporter `:9100`、DCGM `:9400` | 都在 pve02 本机,都在跑 |
+| 依赖的 exporter | ipmi_exporter `:9290`、node_exporter `:9100`、DCGM `:9400` | 自备组件,装法不限,见 2.1 |
### 2.1 前置依赖:三个 exporter(自备组件,装法不限)
@@ -72,12 +72,6 @@ curl -s localhost:9290/metrics | grep ipmi_fan_speed_rpm | head -1
curl -s localhost:9100/metrics | grep Tctl | head -1
```
-> **pve02 当前实况**(仅本机维护参考,不是规定):dcgm-exporter 跑 docker
-> (`nvcr.io/nvidia/k8s/dcgm-exporter:4.6.0-4.8.3-distroless`,`--restart unless-stopped`,
-> `--gpus all`);ipmi_exporter v1.10.1 与 node_exporter 跑 systemd,二进制在
-> `/usr/local/bin/`,node_exporter 带 `--collector.hwmon --collector.cpufreq`,
-> ipmi_exporter 以 root 运行。
-
---
## 3. 首次部署(换机器 / 重装时才需要)
@@ -87,7 +81,7 @@ curl -s localhost:9100/metrics | grep Tctl | head -1
# 它们是控制台的全部数据源,没有这一步控速无依据
# ① 建目录、建 venv
-ssh pve02
+ssh <目标机>
mkdir -p /opt/gpu-fan-console
cd /opt/gpu-fan-console
python3 -m venv .venv
@@ -125,14 +119,17 @@ tar czf /tmp/gfc.tgz \
-C . app
# 3) 传上去
-agentsshcli upload pve02 /tmp/gfc.tgz /root/gfc.tgz # 这条路现在常挂,见「坑 ②」
+agentsshcli upload <连接名> /tmp/gfc.tgz /root/gfc.tgz # 这条路现在常挂,见「坑 ②」
# 4) 远端解压 + 清旧前端产物(⚠️ 见「坑 ③」)
-agentsshcli exec pve02 "tar xzf /root/gfc.tgz -C /opt/gpu-fan-console"
+# 4) 远端解压 + 清旧前端产物(⚠️ 见「坑 ③」)
+# 已配置过的机器升级时排除 config.yaml —— 仓库里的版本是模板(示例地址),
+# 别把现场配好的 prometheus_url 等覆盖回示例值。首次部署才需要带上它。
+agentsshcli exec <连接名> "tar xzf /root/gfc.tgz -C /opt/gpu-fan-console --exclude='app/config.yaml'"
# 5) 重启 + 自检
-agentsshcli exec pve02 "systemctl restart gpu-fan-console && sleep 4 && systemctl is-active gpu-fan-console"
-curl -s http://192.168.6.7:8765/api/status | head -c 300
+agentsshcli exec <连接名> "systemctl restart gpu-fan-console && sleep 4 && systemctl is-active gpu-fan-console"
+curl -s http://192.0.2.10:8765/api/status | head -c 300
```
`deploy.sh` 就是把上面这套串起来,并且**内置了下面三个坑的绕过方案**。
@@ -188,10 +185,10 @@ split -b 30000 all.b64 ch_ # ⚠️ 单次命令行上限 32767 字符
i=0
for f in ch_*; do
[ $i -eq 0 ] && R='>' || R='>>'
- agentsshcli exec pve02 "printf '%s' '$(cat $f)' $R /root/pkg.b64"
+ agentsshcli exec <连接名> "printf '%s' '$(cat $f)' $R /root/pkg.b64"
i=$((i+1))
done
-agentsshcli exec pve02 "base64 -d /root/pkg.b64 > /root/pkg.tgz"
+agentsshcli exec <连接名> "base64 -d /root/pkg.b64 > /root/pkg.tgz"
```
> 只发前端时先 `gzip -9` 再 base64,体积能砍到 1/3,分片数从 25 降到 10。
@@ -214,9 +211,9 @@ find
-type f ! -name '新文件1' ! -name '新文件2' -delete
|---|---|---|
| 服务活着 | `systemctl is-active gpu-fan-console` | `active` |
| 看门狗在跑 | `systemctl list-timers fan-watchdog.timer` | 有下次触发时间 |
-| 页面 200 | `curl -s -o /dev/null -w '%{http_code}' http://192.168.6.7:8765/` | `200` |
-| 静态资源对得上 | `curl -s http://192.168.6.7:8765/ \| grep -o 'index-[A-Za-z0-9_-]*\.\(js\|css\)'` | 与本地 `ls app/static/assets` 一致 |
-| 曲线在控速 | `curl -s http://192.168.6.7:8765/api/status` | 受控位 `duty` 有值、`temperature` 与 GPU 温度对得上 |
+| 页面 200 | `curl -s -o /dev/null -w '%{http_code}' http://192.0.2.10:8765/` | `200` |
+| 静态资源对得上 | `curl -s http://192.0.2.10:8765/ \| grep -o 'index-[A-Za-z0-9_-]*\.\(js\|css\)'` | 与本地 `ls app/static/assets` 一致 |
+| 曲线在控速 | `curl -s http://192.0.2.10:8765/api/status` | 受控位 `duty` 有值、`temperature` 与 GPU 温度对得上 |
| 数据没丢 | 打开设置页 | 模式 / 管控 GPU / 分配关系都是部署前的样子 |
| 启动顺序对 | `journalctl -u gpu-fan-console -n 30` | 先「已应用数据库里的设置」再「已应用数据库里的分配」 |
| 回落动作在 | 同上,搜「安全护栏已激活」 | 有这行才说明退出时会回退 auto |
@@ -232,9 +229,9 @@ git checkout <上一个好提交>
./deploy.sh
# 数据回滚(SQLite 有备份时)
-agentsshcli exec pve02 "systemctl stop gpu-fan-console"
-agentsshcli exec pve02 "cp /root/fan-console.db.bak /opt/gpu-fan-console/app/data/fan-console.db"
-agentsshcli exec pve02 "systemctl start gpu-fan-console"
+agentsshcli exec <连接名> "systemctl stop gpu-fan-console"
+agentsshcli exec <连接名> "cp /root/fan-console.db.bak /opt/gpu-fan-console/app/data/fan-console.db"
+agentsshcli exec <连接名> "systemctl start gpu-fan-console"
```
**最坏情况的保险**:不管服务死没死,都可以直接让 BMC 接管 ——
@@ -242,7 +239,7 @@ agentsshcli exec pve02 "systemctl start gpu-fan-console"
```bash
# 远端执行(把 8 个字节全填 0x00 = 全交回 BMC 自动档)
ipmitool raw 0x3a 0x01 0x00 0x00 0x00 0x00 0x00 0x00 0x00 0x00
-# 宿主都起不来时:从 BMC 独立地址 192.168.6.8 走,凭据 admin/admin
+# 宿主都起不来时:从 BMC 独立管理地址走(BMC 与宿主 OS 相互独立,凭据自行保管,切勿写进公开文档)
```
---
diff --git a/README.md b/README.md
index 59e45d8..1135a10 100644
--- a/README.md
+++ b/README.md
@@ -134,7 +134,7 @@ python -m unittest discover -s app/tests -t . -v
1. **后端单测**:`python -m unittest`(Python 3.13,对齐线上)
2. **前端构建**:`npm ci && npm run build`(自带 vue-tsc 全量类型检查)
3. **部署包**:产出 `gpu-fan-console-app.tgz`(成员路径 `app/...`,排除 `app/data`),
- 挂在 workflow 的 Artifacts 里——下载后传到 pve02 解压重启即可,本机无需装 Node
+ 挂在 workflow 的 Artifacts 里——下载后传到目标机解压重启即可,本机无需装 Node
推送 main 时额外构建**双平台独立可执行文件**(PyInstaller);打 `v*` tag 自动创建
GitHub Release 并附上全部产物:
@@ -201,7 +201,7 @@ alert: # 邮件告警(可选,默认关闭)
max_failed_attempts: 3
email: { ... } # SMTP 配置,支持多收件人、1 小时防轰炸
prometheus:
- base_url: "http://192.168.6.31:30091"
+ base_url: "http://192.0.2.20:30091"
servers:
- type: dell730
ip: "192.168.71.90"
diff --git a/app/__init__.py b/app/__init__.py
index ac3fb04..4467afe 100644
--- a/app/__init__.py
+++ b/app/__init__.py
@@ -5,7 +5,7 @@
2. **API 服务**:REST + WebSocket 给前端
3. **静态托管**:直接伺服前端构建产物
-没有跨机通信、没有独立 agent —— 它只管 pve02 自己这台机器的风扇。
+没有跨机通信、没有独立 agent —— 它只管部署它的这台机器的风扇。
"""
__version__ = "0.1.0"
diff --git a/app/config.py b/app/config.py
index 2b86494..4aae86a 100644
--- a/app/config.py
+++ b/app/config.py
@@ -56,11 +56,11 @@ class SourcesConfig(BaseModel):
# 实时读数**不走**这里,因为 Prometheus 是 30s 快照(见 CLAUDE.md)。
# 但历史曲线恰恰是 Prometheus 的主场,所以单独配一套。
prometheus_url: str | None = None
- #: GPU 温度在 Prometheus 里的 instance 标签,如 ``192.168.6.7:9400``
+ #: GPU 温度在 Prometheus 里的 instance 标签,如 ``192.0.2.10:9400``
prometheus_gpu_instance: str | None = None
- #: CPU 温度(node_exporter 的 hwmon / k10temp)的 instance 标签,如 ``192.168.6.7:9100``
+ #: CPU 温度(node_exporter 的 hwmon / k10temp)的 instance 标签,如 ``192.0.2.10:9100``
prometheus_node_instance: str | None = None
- #: 风扇转速的 instance 标签,如 ``192.168.6.7:9290``
+ #: 风扇转速的 instance 标签,如 ``192.0.2.10:9290``
prometheus_fan_instance: str | None = None
diff --git a/app/config.yaml b/app/config.yaml
index ca46859..0b1100a 100644
--- a/app/config.yaml
+++ b/app/config.yaml
@@ -29,10 +29,11 @@ sources:
http_timeout: 5
# 远程 BMC 兜底(宿主系统起不来时排障用)。默认走本地 in-band,不配这项。
+ # ⚠️ 换成你自己的 BMC 地址与凭据;凭据绝不要提交到公开仓库。
# ipmi_remote:
- # host: "192.168.6.8"
- # user: "admin"
- # password: "admin"
+ # host: "192.0.2.11" # 示例地址(RFC 5737 文档段),按实际替换
+ # user: "your-bmc-user"
+ # password: "your-bmc-password"
# --- 历史趋势(可选,只给 /api/history 用)---
#
@@ -40,10 +41,11 @@ sources:
# 而上面两个 endpoint 是直连 exporter,读的才是当下值。
# 但历史曲线恰恰是 Prometheus 的主场,所以单独配一套。
# 不配的话界面上的趋势图会显示"暂无历史数据",其余功能不受影响。
- prometheus_url: "http://192.168.6.31:30091"
- prometheus_gpu_instance: "192.168.6.7:9400" # GPU 温度 ← DCGM exporter
- prometheus_node_instance: "192.168.6.7:9100" # CPU 温度 ← node_exporter
- prometheus_fan_instance: "192.168.6.7:9290" # 风扇转速 ← ipmi_exporter
+ # ⚠️ 下面的地址全部是示例(RFC 5737 文档段),按你的实际环境替换。
+ prometheus_url: "http://192.0.2.20:30091" # 示例:你的 Prometheus 地址
+ prometheus_gpu_instance: "192.0.2.10:9400" # GPU 温度 ← DCGM exporter
+ prometheus_node_instance: "192.0.2.10:9100" # CPU 温度 ← node_exporter
+ prometheus_fan_instance: "192.0.2.10:9290" # 风扇转速 ← ipmi_exporter
curve:
# 滞回带(°C):降温方向必须跌出这个带宽才降档,防止温度在阈值附近
diff --git a/app/ipmi.py b/app/ipmi.py
index eabbdcf..568d40c 100644
--- a/app/ipmi.py
+++ b/app/ipmi.py
@@ -23,8 +23,8 @@
b1 CPU1_FAN1
b2 --(保留)
b3 REAR_FAN1
- b4 REAR_FAN2 ← pve02 用于 Tesla T10 散热
- b5 FRNT_FAN1 ← pve02 用于 Tesla T10 散热
+ b4 REAR_FAN2 ← 宿主机用于 Tesla T10 散热
+ b5 FRNT_FAN1 ← 宿主机用于 Tesla T10 散热
b6 FRNT_FAN2 (未接风扇)
b7 FRNT_FAN3 (未接风扇)
b8 FRNT_FAN4 (未接风扇)
@@ -148,9 +148,9 @@ def encode_duty(duty: int | None) -> int:
class IPMIClient:
"""ipmitool 封装。
- 默认走**本地 in-band**(``/dev/ipmi0``,需要 root),这是 pve02 上的推荐用法:
+ 默认走**本地 in-band**(``/dev/ipmi0``,需要 root),这是推荐用法:
链路最短、无网络依赖。也支持 ``lanplus`` 远程模式指向 BMC 独立地址
- (``192.168.6.8``),用于宿主系统起不来时的带外兜底。
+ (``192.0.2.11``),用于宿主系统起不来时的带外兜底。
"""
def __init__(
diff --git a/app/main.py b/app/main.py
index 6ad250d..27bdcb4 100644
--- a/app/main.py
+++ b/app/main.py
@@ -186,7 +186,7 @@ async def lifespan(app: FastAPI):
app = FastAPI(
title="GPU Fan Console",
- description="pve02 GPU 温度联动风扇控制台(单机应用)",
+ description="GPU 温度联动风扇控制台(单机应用)",
version=__version__,
lifespan=lifespan,
)
diff --git a/app/sensors.py b/app/sensors.py
index 1c4fd30..16c0e26 100644
--- a/app/sensors.py
+++ b/app/sensors.py
@@ -13,7 +13,7 @@
- 所有网络/子进程调用**带超时**。上游项目的教训:一个不带超时的阻塞读
能把整个控制线程静默挂死。
-关于 pve02 这块板子(ASRock Rack EPYCD8,BMC 固件 2.20)的实测要点:
+关于本机这块板子(ASRock Rack EPYCD8,BMC 固件 2.20)的实测要点:
- ``ipmitool sdr type fan`` 输出**五列**:``名称 | 传感器ID | 状态 | 阈值 | 读数``。
读数在**最后一列**。最初按「第二列是读数」写会把传感器 ID ``62h`` 当成 RPM。
diff --git a/app/tests/test_core.py b/app/tests/test_core.py
index c71e6de..0c7ef80 100644
--- a/app/tests/test_core.py
+++ b/app/tests/test_core.py
@@ -338,7 +338,7 @@ def test_emergency_thresholds_accept_valid_pair(self) -> None:
class TestPrometheusParsing(unittest.TestCase):
- """Prometheus 文本解析 —— 用 pve02 上抓到的真实格式。"""
+ """Prometheus 文本解析 —— 用 真实设备抓到的格式。"""
SAMPLE = """\
# HELP DCGM_FI_DEV_GPU_TEMP GPU temperature (in C).
@@ -428,7 +428,7 @@ def test_missing_clock_stays_none(self) -> None:
class TestSdrFanParsing(unittest.TestCase):
"""``ipmitool sdr type fan`` 输出解析。
- 样例取自 2026-09-28 在 pve02 上的真实输出 —— 这组用例的由来就是一个
+ 样例取自 2026-09-28 在真实设备上的输出 —— 这组用例的由来就是一个
真实 bug:最初以为「第二列是读数」,结果把传感器 ID ``62h`` 当成了 RPM。
"""
@@ -499,7 +499,7 @@ def test_no_reading_temperature(self) -> None:
class TestFanReaderIpmitoolFallback(unittest.TestCase):
"""ipmitool 兜底路径 —— 必须跳过未接的风扇位。
- 样例是 2026-09-28 在 pve02 上抓的真实输出:14 个风扇传感器位里只有
+ 样例是 2026-09-28 在真实设备上抓的输出:14 个风扇传感器位里只有
4 个有读数,其余全是 ``No Reading``。不跳过的话界面上会凭空多出
10 个空风扇位。
"""
diff --git a/deploy.sh b/deploy.sh
index 20302fa..ec0f178 100644
--- a/deploy.sh
+++ b/deploy.sh
@@ -1,24 +1,24 @@
#!/usr/bin/env bash
#
-# 一键部署 GPU 风扇控制台到 pve02
+# 一键部署 GPU 风扇控制台到目标机
#
# ./deploy.sh 全量(后端 + 前端)→ 覆盖 → 重启 → 自检
# ./deploy.sh --static-only 只更新前端(不改后端,不重启,静态文件即时生效)
# ./deploy.sh --no-build 跳过前端构建(复用 app/static 里已有的产物)
# ./deploy.sh --no-restart 传完不重启(全量模式下慎用)
#
-# 可用环境变量覆盖默认值:
-# CONN=pve02 REMOTE_DIR=/opt/gpu-fan-console SERVICE=gpu-fan-console URL=http://192.168.6.7:8765
+# 可用环境变量(CONN 必填,其余有默认值):
+# CONN= REMOTE_DIR=/opt/gpu-fan-console SERVICE=gpu-fan-console URL=http://<目标机>:8765
#
# 为什么不用 agentsshcli upload:它现在报「创建远端续传元数据失败」,虽然数据传完了
# 但整体返回失败(--no-cache 直连模式也无效,已实测)。所以走分片 base64,见 DEPLOY.md 坑 ②。
#
set -euo pipefail
-CONN="${CONN:-pve02}"
+CONN="${CONN:?用法: CONN= ./deploy.sh}"
REMOTE_DIR="${REMOTE_DIR:-/opt/gpu-fan-console}"
SERVICE="${SERVICE:-gpu-fan-console}"
-URL="${URL:-http://192.168.6.7:8765}"
+URL="${URL:-http://127.0.0.1:8765}"
# 单次命令行上限 32767 字符(Windows),留出余量
CHUNK_SIZE="${CHUNK_SIZE:-30000}"
@@ -71,8 +71,11 @@ if [ "$STATIC_ONLY" -eq 1 ]; then
tar czf "$PKG" -C . app/static
else
# ⚠️ --exclude='app/data' 是在保命:那是 SQLite 权威数据源,被覆盖等于丢失全部配置
+ # ⚠️ --exclude='app/config.yaml':仓库里的是模板(示例地址),覆盖会毁掉现场配好的
+ # prometheus_url 等本机值。首次部署请手动传一次完整 config.yaml 再改。
tar czf "$PKG" \
--exclude='__pycache__' --exclude='*.pyc' --exclude='app/data' \
+ --exclude='app/config.yaml' \
-C . app
fi
echo " 包大小:$(wc -c <"$PKG") bytes"
diff --git a/fanController/base_controller.py b/fanController/base_controller.py
index 5ad4d28..497fe5b 100644
--- a/fanController/base_controller.py
+++ b/fanController/base_controller.py
@@ -21,7 +21,7 @@ def __init__(self, servers, interval, windows_ipmi_tool_path, logger, auto=True,
auto (bool): 是否自动模式,True为自动模式,False为手动模式。
alert_config (dict, optional): 告警配置字典。
prometheus_config (dict, optional): Prometheus 查询配置,形如
- ``{'base_url': 'http://192.168.6.31:30091', 'instance': '192.168.6.7:9290'}``。
+ ``{'base_url': 'http://192.0.2.20:30091', 'instance': '192.0.2.10:9290'}``。
风扇转速统一从这里取(见 :meth:`get_fan_rotational_speed`)。
"""
self.platform_system = platform.system()
@@ -130,11 +130,11 @@ def _pick_instance(self, *keys):
⚠️ **不同数据源的 instance 是不同的**,因为它们是各自的 exporter:
============ ================== ==========================
- 数据 指标 instance(pve02 实测)
+ 数据 指标 instance(实测示例)
============ ================== ==========================
- 风扇转速 ipmi_fan_speed_rpm ``192.168.6.7:9290``
- CPU 温度 node_hwmon_temp_* ``192.168.6.7:9100``
- GPU 温度 DCGM_FI_DEV_* ``192.168.6.7:9400``
+ 风扇转速 ipmi_fan_speed_rpm ``192.0.2.10:9290``
+ CPU 温度 node_hwmon_temp_* ``192.0.2.10:9100``
+ GPU 温度 DCGM_FI_DEV_* ``192.0.2.10:9400``
============ ================== ==========================
混用一个 ``instance`` 会直接查不到数据 —— 这个坑 2026-09-28 实现时踩到。
diff --git a/fanController/epycd8_controller.py b/fanController/epycd8_controller.py
index ad88be2..fec7042 100644
--- a/fanController/epycd8_controller.py
+++ b/fanController/epycd8_controller.py
@@ -27,11 +27,11 @@
b1 CPU1_FAN1
b2 --(保留)
b3 REAR_FAN1
- b4 REAR_FAN2 ← pve02 用于 Tesla T10 散热
- b5 FRNT_FAN1 ← pve02 用于 Tesla T10 散热
- b6 FRNT_FAN2 (pve02 未接风扇)
- b7 FRNT_FAN3 (pve02 未接风扇)
- b8 FRNT_FAN4 (pve02 未接风扇)
+ b4 REAR_FAN2 ← 宿主机用于 Tesla T10 散热
+ b5 FRNT_FAN1 ← 宿主机用于 Tesla T10 散热
+ b6 FRNT_FAN2 (未接风扇)
+ b7 FRNT_FAN3 (未接风扇)
+ b8 FRNT_FAN4 (未接风扇)
"""
from .base_controller import IPMIFanController
@@ -94,7 +94,7 @@ def _build_temperature_query(self):
字段名,在本机型上承载的其实是 GPU 温度。
数据源同样是 Prometheus —— DCGM exporter 已接入
- (job ``dcgm-exporter-pve02``),指标 ``DCGM_FI_DEV_GPU_TEMP``。
+ (job 名自定义,如 ``dcgm-exporter``),指标 ``DCGM_FI_DEV_GPU_TEMP``。
Returns:
str | None: PromQL;未配置 ``gpu_instance`` 时返回 ``None``。
@@ -103,7 +103,7 @@ def _build_temperature_query(self):
if not instance:
self.logger.error(
f"服务器 {self.ip}: 未配置 prometheus.gpu_instance,无法定位 "
- f"DCGM 数据(例: 192.168.6.7:9400)"
+ f"DCGM 数据(例: 192.0.2.10:9400)"
)
return None
return f'DCGM_FI_DEV_GPU_TEMP{{instance="{instance}"}}'
diff --git a/frontend/src/components/Sidebar.vue b/frontend/src/components/Sidebar.vue
index 6769f65..5aa9114 100644
--- a/frontend/src/components/Sidebar.vue
+++ b/frontend/src/components/Sidebar.vue
@@ -14,6 +14,9 @@ defineProps<{
const emit = defineEmits<{ (e: 'navigate', p: PageKey): void }>()
+// 同源部署:直接显示浏览器当前访问的地址,不写死任何主机名/IP
+const accessAddress = window.location.host || '同源部署'
+
const NAV: { key: PageKey; label: string; hint: string }[] = [
{ key: 'dashboard', label: '概览', hint: 'GPU 状态 · 温度 · 趋势' },
{ key: 'fans', label: '风扇控制', hint: '分配 · 调速 · 曲线' },
@@ -38,7 +41,7 @@ const NAV: { key: PageKey; label: string; hint: string }[] = [
GPU 风扇控制台
-
pve02 · 2× Tesla T10
+
2× Tesla T10
@@ -123,7 +126,7 @@ const NAV: { key: PageKey; label: string; hint: string }[] = [
/>
{{ connected ? '实时连接' : '轮询模式' }}
- 192.168.6.7:8765
+ {{ accessAddress }}