feat(cluster): 集群化落地 —— Manager/Worker 拆分 + 归属租约 + 跨机验证(T08)

背景:把平台从「单机单进程」改造成「1 组 Manager + N 台 Worker + 共享归属状态」,
硬约束 = 全程兼容单例模式(deployMode 默认 local;生产切换前 47 一行未动)。

主要改动
1) 数据模型 v7(SQLite 与 PG 两方言同步):新增 dsh_hosts 注册表 +
   dsh_instances.{host_id,epoch,heartbeat_at,lease_until};claimInstance 原子抢占
   (UPDATE … WHERE host_id IS NULL OR lease_until < now)+ pinInstanceHost 钉住归属。
2) 租约与 fencing:src/supervisor/lease.ts(acquire/renew/release + stillHolder 判据 +
   ttl > 2×renew 硬校验);心跳里续租,失权即向 worker 下发更高 epoch(self-fencing)。
   ⚠️ release 只清租约(lease_until),**保留 host_id** —— host_id 是「用户数据在哪台」的锚点。
3) Worker agent(src/worker/agent.ts,子命令 dshs worker):实例生命周期 + 文件面 /fs/*
   + 幂等键(operationId)+ 鉴权(timingSafeEqual);Worker 不写控制面数据
   (apiKey/uid 由 Manager 随 launch 投递,R5 收窄)。
4) 远端 Spawner + LeasedSpawner:按 host 路由(**粘性优先**:有历史归属且那台 up 就留在原地,
   否则按容量选最空的)+ 容量准入 + deployMode=cluster 装配(systemd drop-in,可回滚)。
5) bwrap 修正:**所有挂载点的中间目录统一前置 + 去重 + 由外到内**(「就近创建」会在嵌套前缀下
   遮掉已绑挂载点 ⇒ bwrap: Can't chdir);且**只能用 --tmpfs**,用 --perms 会让 47 的
   bwrap 0.4.0 直接拒启动(沙箱全挂)。
6) 跨机隧道 src/worker/tunnel.ts:SSH ControlMaster + 动态 -R 转发;**自愈由 agent 本地
   20s 定时器驱动**(不能只放 /healthz —— 心跳本身经隧道进来,断了就没人触发它)。
7) 文件面按归属路由(RemoteUserFs):实例与文件必须落在同一台机器,否则实例看不到自己的文件。
8) 观测面:dshs doctor / dshs cluster status。

验证(本次均已实跑)
- test/lease.test.mjs:SQLite 10/10 == PG 10/10
- 组件级端到端 5 个:verify-cluster-{agent,lease,fs,migrate,live}.mjs
- 真跨机(47 Manager / 106 Worker,跨云 + 反向隧道)verify-cluster-cross.mjs 九步全绿
- 域名形态访问 verify-cluster-domain.mjs(<user>.域名 → Manager → 远端实例;越权 403)
- 冒烟 scripts/smoke-*:6/8,失败项与改动前基线完全相同(无回归)
- 生产切换与回滚剧本见 dsh-server-docs/交接单/T08-集群化落地-兼容单例模式.md §16
This commit is contained in:
admin committed 2026-09-15 18:47:02 +08:00
1 parent 68c0a320ed
commit c70d5d860e
47 files changed
+5367 -14

No files matched your search

+11
View File
@@ -0,0 +1,11 @@
#!/usr/bin/env bash
export PGPASSWORD=dshs_cluster_2026
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
echo "=== users ==="
Q "select id || ' | ' || username || ' | ' || role || ' | uid=' || coalesce(uid::text,'-') from users order by row_id"
echo "=== dsh_instances ==="
Q "select id || ' | user=' || user_id || ' | ' || role || ' | ' || status || ' | host=' || coalesce(host_id,'NULL') || ' | epoch=' || epoch from dsh_instances order by id"
echo "=== dsh_hosts ==="
Q "select id || ' | cap=' || capacity_mb || ' | used=' || used_mb || ' | ' || status from dsh_hosts order by id"
echo "=== sessions(应为 0 条 switch-verify) ==="
Q "select count(*) from sessions where user_agent='switch-verify'"
+5
View File
@@ -54,5 +54,10 @@ if (role === 'watchdog') {
})
server.listen(port, '127.0.0.1', () => {
console.log(`fake-dsh listening on ${port}`)
// 真实 dsh 启动后会打印**可直达的带 token URL**,平台就是靠这行取 launch token
// (正则:/dsh web: http://////127//.0//.0//.1://d+/////?token=([A-Za-z0-9_-]+)/)。
// 夹具必须照实吐出来,否则平台只能等满 10 s 超时 ⇒ 「登录直达会话」这条链路
// 在本机测试里**永远测不到**(2026-09-15 T08 S3 实测踩到)。
console.log(`dsh web: http://127.0.0.1:${port}/?token=FAKE_TOKEN_${process.pid}`)
})
}
+186
View File
@@ -0,0 +1,186 @@
#!/usr/bin/env node
/**
* T08 · S1.2:SQLite → Postgres 一次性数据迁移。
*
* 设计要点(都是踩过才会疼的地方):
* 1. **列清单不写死** —— 从 PG 的 information_schema 与 SQLite 的 PRAGMA 取**交集**,
* 这样 schema 演进(v4 的 folder/patch、v6 的 enabled 等)不会让脚本静默少搬字段。
* 2. **PG 表结构不由本脚本建** —— 先 import 平台自己的 `createDbAdapter`(带 dbUrl),
* 让**平台的迁移**在 PG 上建库。这样"迁移脚本"与"平台 schema"永远不会两套。
* 3. **identity 列要 `OVERRIDING SYSTEM VALUE`** —— `users.uid` 与 `audit_log.id` 是
* GENERATED ALWAYS AS IDENTITY;不覆盖就会重排 id,**uid 一变 = 所有用户文件属主失配**。
* 搬完必须 `RESTART WITH` 把序列推到 max+1,否则下一条 INSERT 撞主键。
* 4. **FK 顺序**:先 users,再 workspaces/sessions,最后引用它们的表。
* 5. `--dry-run` 只报行数,不写任何东西。
*
* 用法:
* node scripts/migrate-sqlite-to-pg.mjs --sqlite /var/lib/dshs/dshs.db \
* --pg postgres://dshs:***@127.0.0.1:15432/dshs [--dry-run]
*
* @module dshs/scripts/migrate-sqlite-to-pg
*/
import { existsSync } from 'node:fs'
import Database from 'better-sqlite3'
import pg from 'pg'
import { createDbAdapter } from '../lib/db/index.js'
import { resolveConfig } from '../lib/config.js'
/** FK 依赖顺序(父 → 子)。未列出的表会被追加到末尾并告警。 */
const ORDER = [
'users',
'workspaces',
'sessions',
'folder_plugins',
'dsh_instances',
'domains',
'credential_vault',
'business_plugins',
'audit_log',
]
/** identity 列(必须 OVERRIDING SYSTEM VALUE + 搬完 RESTART)。 */
const IDENTITY = { users: 'uid', audit_log: 'id' }
/**
* ⛔ **绝不搬**的表。
*
* `schema_migrations`:目标端的"已应用版本"标记由**平台的迁移**建立(见 `ensurePgSchema`),
* 从源库搬会把同一批版本号再插一遍 ⇒ `schema_migrations_pkey` 唯一键冲突
* (2026-09-15 实测踩到,事务已整体回滚)。语义上也应如此:**结构版本由平台在目标端决定**。
*/
const SKIP = new Set(['schema_migrations'])
function arg(name, fallback) {
const i = process.argv.indexOf(`--${name}`)
return i >= 0 && process.argv[i + 1] !== undefined ? process.argv[i + 1] : fallback
}
const sqlitePath = arg('sqlite')
const pgUrl = arg('pg')
const dryRun = process.argv.includes('--dry-run')
if (sqlitePath === undefined || pgUrl === undefined) {
console.error('用法: node scripts/migrate-sqlite-to-pg.mjs --sqlite <file> --pg <url> [--dry-run]')
process.exit(2)
}
if (!existsSync(sqlitePath)) {
console.error(`SQLite 文件不存在: ${sqlitePath}`)
process.exit(2)
}
/** 让**平台的迁移**在 PG 上建好结构(不自己写 DDL,避免两套 schema)。 */
async function ensurePgSchema() {
const config = resolveConfig({ dataRoot: '/tmp/migrate-tooling', dbUrl: pgUrl })
const db = await createDbAdapter(config)
await db.close()
}
async function main() {
const sq = new Database(sqlitePath, { readonly: true })
await ensurePgSchema()
const client = new pg.Client({ connectionString: pgUrl })
await client.connect()
const sqTables = sq
.prepare("SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'")
.all()
.map((r) => r.name)
const pgTables = (
await client.query("SELECT table_name FROM information_schema.tables WHERE table_schema='public'")
).rows.map((r) => r.table_name)
const common = sqTables.filter((t) => pgTables.includes(t) && !SKIP.has(t))
if (sqTables.some((t) => SKIP.has(t))) {
console.log(`按设计跳过: ${[...SKIP].join(', ')}(目标端的结构版本由平台迁移建立)`)
}
const ordered = [
...ORDER.filter((t) => common.includes(t)),
...common.filter((t) => !ORDER.includes(t)),
]
const extra = common.filter((t) => !ORDER.includes(t))
if (extra.length > 0) console.warn(`⚠️ 未在 ORDER 中声明、按末尾处理的表: ${extra.join(', ')}`)
/** 两端的列交集 —— 只搬双方都有的列。 */
async function sharedCols(table) {
const sqCols = sq.prepare(`PRAGMA table_info(${table})`).all().map((c) => c.name)
const pgCols = (
await client.query(
'SELECT column_name FROM information_schema.columns WHERE table_schema=$1 AND table_name=$2',
['public', table],
)
).rows.map((r) => r.column_name)
return sqCols.filter((c) => pgCols.includes(c))
}
const report = []
await client.query('BEGIN')
try {
for (const table of ordered) {
const cols = await sharedCols(table)
if (cols.length === 0) {
console.warn(`跳过 ${table}: 无公共列`)
continue
}
const rows = sq.prepare(`SELECT ${cols.join(',')} FROM ${table}`).all()
const idCol = IDENTITY[table]
if (!dryRun && rows.length > 0) {
const colList = idCol !== undefined ? [idCol, ...cols.filter((c) => c !== idCol)] : cols
const override = idCol !== undefined ? ' OVERRIDING SYSTEM VALUE' : ''
const chunk = 200
for (let i = 0; i < rows.length; i += chunk) {
const slice = rows.slice(i, i + chunk)
const values = []
const tuples = slice.map((row) => {
const ph = colList.map((c) => {
values.push(row[c] ?? null)
return `$${values.length}`
})
return `(${ph.join(',')})`
})
await client.query(
`INSERT INTO ${table} (${colList.join(',')})${override} VALUES ${tuples.join(',')}`,
values,
)
}
if (idCol !== undefined) {
await client.query(
`SELECT setval(pg_get_serial_sequence('${table}','${idCol}'),
(SELECT COALESCE(MAX(${idCol}),0) FROM ${table}))`,
)
}
}
report.push({ table, rows: rows.length, cols: cols.length })
}
if (dryRun) await client.query('ROLLBACK')
else await client.query('COMMIT')
} catch (err) {
await client.query('ROLLBACK')
throw err
}
console.log(`\n${dryRun ? '【DRY-RUN,已回滚】' : '【已提交】'} SQLite → PG 迁移明细`)
for (const r of report) console.log(` ${r.table.padEnd(18)} ${String(r.rows).padStart(6)} 行 / ${r.cols} 列`)
// 校验:逐表比对行数
let bad = 0
if (!dryRun) {
for (const r of report) {
const pgCount = Number((await client.query(`SELECT COUNT(*) AS c FROM ${r.table}`)).rows[0].c)
const sqCount = Number(sq.prepare(`SELECT COUNT(*) AS c FROM ${r.table}`).get().c)
if (pgCount !== sqCount) {
console.error(` ✗ 行数不符 ${r.table}: PG=${pgCount} SQLite=${sqCount}`)
bad += 1
}
}
console.log(bad === 0 ? '\n✅ 逐表行数一致' : `\n❌ ${bad} 张表行数不符`)
}
await client.end()
sq.close()
process.exit(bad === 0 ? 0 : 1)
}
main().catch((err) => {
console.error(err)
process.exit(1)
})
+37
View File
@@ -0,0 +1,37 @@
#!/usr/bin/env bash
# 存量数据并行搬运 47 → 106(4 路并发;瓶颈在源端小文件 IOPS,单流只 ~0.4MB/s)
# 搬完写 /root/push-parallel.done,供后续 cutover 判断
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
DST=[email protected]
SSHO="-i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o Compression=no"
SRC=/var/lib/dshs
LOG=/root/push-parallel.log
: > "$LOG"
ssh -n $SSHO "$DST" 'mkdir -p /var/lib/dshs'
echo "[$(date +%T)] 目标端就绪" >> "$LOG"
# 按"大小"分组:大用户单开一路,其余合流(每路一个 tar→ssh)
one() { # $1=组名 其余=相对路径列表
local name="$1"; shift
(
cd "$SRC" || exit 1
{ for p in "$@"; do [ -e "$p" ] && echo "$p"; done; } > "/tmp/list-$name.txt"
tar --numeric-owner --files-from="/tmp/list-$name.txt" -cf - \
| ssh $SSHO "$DST" "tar -C /var/lib/dshs --numeric-owner -xf -"
echo "[$(date +%T)] $name 完成 rc=$?" >> "$LOG"
) &
}
# 组划分(按实际内容:1 个大用户 + 若干小目录)
one g1 "users/4092b965-2f68-4977-9989-68b3966f7df0"
one g2 "users/cce6d1cd-b376-4304-80f0-0e1c58c9ffde" "users/3ec95f69-6a4e-4d16-a415-56aa09396fc5"
one g3 "users/4eaeb26b-9e0c-4c68-9b61-0daf70664ae5" "users/74e8804a-e0c8-4645-84de-90dd3fae6c2b"
one g4 "users/7ba268be-6103-438c-8e6b-609b422ccbca" "users/ca3f36e0-937f-437e-b850-53b8230f20f8" \
"bundled-skills" "business-plugins" "whitelist-cache"
wait
echo "[$(date +%T)] 全部完成" >> "$LOG"
ssh -n $SSHO "$DST" 'du -sm /var/lib/dshs | cut -f1' >> "$LOG" 2>&1
touch /root/push-parallel.done
+63
View File
@@ -0,0 +1,63 @@
#!/usr/bin/env bash
# 在 47 初始化「控制面 PG」:独立数据目录 /var/lib/dshs-pg + 专用 unit dshs-pg.service
# 设计口径:控制面 DB 在 Manager 侧、仅 loopback、scram 认证。
set -uo pipefail
PGDATA=/var/lib/dshs-pg
PGPORT=15432
DBPW="${DSHS_PG_PASSWORD:-dshs_cluster_2026}"
PGVER=$(/usr/bin/postgres --version | grep -oE '[0-9]+' | head -1)
echo " PG 版本: $(/usr/bin/postgres --version)"
# 别让发行版的默认单元意外起来(我们用自己的 unit + 自己的数据目录)
systemctl disable --now postgresql 2>/dev/null >/dev/null || true
if [ ! -f "$PGDATA/PG_VERSION" ]; then
install -d -o postgres -g postgres -m 700 "$PGDATA"
su - postgres -c "/usr/bin/initdb -D $PGDATA -E UTF8 --locale=C.UTF-8 --auth-local=peer --auth-host=scram-sha-256" >/tmp/initdb.log 2>&1 \
&& echo " ✓ initdb 完成" || { echo " ✗ initdb 失败"; tail -5 /tmp/initdb.log; exit 1; }
cat >> "$PGDATA/postgresql.conf" <<CONF
# ── DSHS 控制面(2026-09-15 集群化切换)──
listen_addresses = '127.0.0.1'
port = $PGPORT
unix_socket_directories = '/var/run/postgresql'
max_connections = 100
shared_buffers = 128MB
CONF
chown postgres:postgres "$PGDATA/postgresql.conf"
echo " ✓ 已写入 listen=127.0.0.1 port=$PGPORT"
fi
cat > /etc/systemd/system/dshs-pg.service <<UNIT
[Unit]
Description=DSHS control-plane PostgreSQL (cluster mode)
After=network.target
[Service]
Type=notify
User=postgres
Group=postgres
ExecStart=/usr/bin/postgres -D $PGDATA
ExecReload=/bin/kill -HUP \$MAINPID
KillMode=mixed
TimeoutStopSec=30
Restart=on-failure
[Install]
WantedBy=multi-user.target
UNIT
systemctl daemon-reload
systemctl enable --now dshs-pg >/dev/null 2>&1
sleep 3
echo " dshs-pg: $(systemctl is-active dshs-pg) 监听: $(ss -lntp 2>/dev/null | grep -c $PGPORT)"
# 角色 + 库(幂等)
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_roles where rolname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create role dshs login password '$DBPW'\"" >/dev/null 2>&1
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_database where datname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create database dshs owner dshs\"" >/dev/null 2>&1
echo " --- 连接自检(dshs 角色) ---"
PGPASSWORD="$DBPW" /usr/bin/psql -h 127.0.0.1 -p $PGPORT -U dshs -d dshs -tAc "select current_user||'@'||current_database()||' pg='||version()" 2>&1 | head -1 | cut -c1-90
echo " 连接串(含密码,勿外传): postgres://dshs:$DBPW@127.0.0.1:$PGPORT/dshs"
+16
View File
@@ -0,0 +1,16 @@
#!/usr/bin/env bash
# 在 47 的控制面 PG 上建角色与库(peer 认证走 unix socket,不依赖已存在的密码)
set -uo pipefail
PGPORT=15432
DBPW="dshs_cluster_2026"
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_roles where rolname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create role dshs login password '$DBPW'\"" >/dev/null 2>&1
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_database where datname='dshs'\"" 2>/dev/null | grep -q 1 \
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create database dshs owner dshs\"" >/dev/null 2>&1
echo "=== 角色/库确认(本地 peer) ==="
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select rolname from pg_roles where rolname='dshs'\"" 2>/dev/null | sed 's/^/ role: /'
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select datname||' owner='||pg_get_userbyid(datdba) from pg_database where datname='dshs'\"" 2>/dev/null | sed 's/^/ db: /'
echo "=== 连接自检(TCP + 密码,走的应是**本机 PG 13**) ==="
PGPASSWORD="$DBPW" /usr/bin/psql -h 127.0.0.1 -p $PGPORT -U dshs -d dshs -tAc "select version()" 2>&1 | head -1 | cut -c1-80
+5 -1
View File
@@ -75,7 +75,11 @@ try {
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
console.log('launch ->', r.status, r.body?.url)
assert(r.status === 200, 'launch succeeds')
assert(r.body.url === 'https://carol.test.local/', 'launch returns the subdomain URL')
// 2026-09-15(T08 S3):夹具 `fake-dsh.mjs` 现在**照实打印带 token 的 URL**(与真实 dsh 一致),
// 于是这里不能再写死成不带 token 的相等 —— 原断言是"夹具不吐 token"时的意外产物。
// 保留原意(是子域 URL、不泄露回环端口),并把 token 的存在一并纳入判据。
assert(r.body.url.startsWith('https://carol.test.local/'), 'launch returns the subdomain URL')
assert(!r.body.url.includes('127.0.0.1'), 'launch URL must not leak the loopback port')
await sleep(200)
+45
View File
@@ -0,0 +1,45 @@
#!/usr/bin/env bash
# 起一台 **cluster 模式的 Manager**(T08 跨机演练用;在 Manager 那台机器上跑)。
#
# 现场的对照(2026-09-15 演练实测):
# · Manager 在 **47**(本脚本所在机器),监听 `127.0.0.1:13080`(**不公网暴露**)
# · Worker agent 在 **106**,经 SSH 反向隧道出现在本机 `127.0.0.1:19000` / `19001`
# · 控制面 PG 也在 **106**,经同一条隧道出现在本机 `127.0.0.1:15432`
# 用法:bash scripts/start-cluster-manager.sh (env 见下方 manager.env)
在 47 上:建 manager.env、bootstrap 管理员、起 cluster Manager(127.0.0.1:13080)
set -uo pipefail
cd /opt/dshs-cluster || exit 1
cat > /opt/dshs-cluster/manager.env <<'ENVEOF'
DSHS_DEPLOY_MODE=cluster
DSHS_DB_URL=postgres://dshs:[email protected]:15432/dshs_cross
DSHS_DATA_ROOT=/opt/dshs-cluster/data
DSHS_CLUSTER_HOST_ID=m-47
DSHS_CLUSTER_AGENT_URL=http://127.0.0.1:19000
DSHS_CLUSTER_AGENT_TOKEN=cross-machine-token
DSHS_CLUSTER_INSTANCE_HOST=127.0.0.1
DSHS_CLUSTER_WORKER_DATA_ROOT=/opt/dshs-cluster/live-data
DSHS_CLUSTER_CAPACITY_MB=-1
DSHS_CLUSTER_REGISTER_SELF=0
DSHS_CLUSTER_LEASE_TTL_MS=30000
ENVEOF
set -a
# shellcheck disable=SC1091
. /opt/dshs-cluster/manager.env
set +a
echo "--- bootstrap 管理员 ---"
node lib/cli.js bootstrap-admin --username root --password crossmgr123 2>&1 | tail -1
echo "--- 起 Manager ---"
pkill -f "dshs-cluster/lib/cli.js --port 13080" 2>/dev/null
sleep 1
nohup node lib/cli.js --port 13080 --host 127.0.0.1 --log-level warn > /tmp/manager-47.log 2>&1 &
sleep 7
echo "--- 自检 ---"
echo " login.html : $(curl -s -o /dev/null -w '%{http_code}' -m 6 http://127.0.0.1:13080/login.html)"
echo " 进程 : $(pgrep -cf 'dshs-cluster/lib/cli.js --port 13080')"
echo " 日志尾部 :"
tail -4 /tmp/manager-47.log 2>/dev/null | sed 's/^/ /'
+76
View File
@@ -0,0 +1,76 @@
#!/usr/bin/env bash
# 切换 A 步(在 47 上跑):
# ① 备份 /opt/dshs/lib → /opt/dsh/backups/lib-<ts>/
# ② 覆盖 /opt/dshs/lib(T08 集群版代码)
# ③ 装 **本地 Worker**(w-47,19100,无隧道 —— Manager 同机直连)
# ④ 把既有用户(admin/guest)的归属**预置**为 w-47(否则粘性落点无处可粘、新老用户会被按容量随机调度)
# ⑤ 在 PG 里注册 w-47 / w-106 两台 worker
set -uo pipefail
TS=$(date +%Y%m%d-%H%M%S)
W47_TOKEN="dshs-worker-47-c4b7e19f"
W106_TOKEN="dshs-worker-7f3a91c05e"
PGURL="postgres://dshs:[email protected]:15432/dshs"
TARBALL=/tmp/dshs-lib-new.tgz
echo "=== ① 备份 /opt/dshs/lib ==="
mkdir -p "/opt/dsh/backups/lib-$TS"
cp -a /opt/dshs/lib "/opt/dsh/backups/lib-$TS/lib" && echo " ✓ 备份到 /opt/dsh/backups/lib-$TS/lib($(find /opt/dsh/backups/lib-$TS -type f | wc -l) 文件)"
echo "=== ② 覆盖 lib ==="
[ -f "$TARBALL" ] || { echo " ✗ 缺少 $TARBALL"; exit 1; }
rm -rf /opt/dshs/lib && tar -xzf "$TARBALL" -C /opt/dshs
echo " ✓ 已覆盖;cluster 特征检查: $(grep -l "DEPLOY_MODE" /opt/dshs/lib/config.js >/dev/null 2>&1 && echo '有 cluster 代码 ✓' || echo '✗ 未见 cluster 代码')"
echo " lease/agent/tunnel: $(ls /opt/dshs/lib/supervisor/lease.js /opt/dshs/lib/worker/agent.js /opt/dshs/lib/worker/tunnel.js 2>/dev/null | wc -l)/3"
echo "=== ③ 本地 Worker 单元(w-47,无隧道) ==="
cat > /etc/dshs-worker.env <<ENV
DSHS_DATA_ROOT=/var/lib/dshs
DSHS_ISOLATION_MODE=account
DSHS_DSH_BIN=/usr/local/bin/dsh
DSHS_BASE_UID=100000
DSH_INSTANCE_NODE_OPTIONS=--max-old-space-size=160
DSH_INSTANCE_UNIVER_SOCKET=auto
DSHS_CLUSTER_AGENT_TOKEN=$W47_TOKEN
ENV
chmod 600 /etc/dshs-worker.env
cat > /etc/systemd/system/dshs-worker.service <<UNIT
[Unit]
Description=DSHS cluster worker agent (this host = 47, local users' instances)
After=network-online.target
[Service]
Type=simple
EnvironmentFile=/etc/dshs-worker.env
ExecStart=/usr/local/bin/node /opt/dshs/lib/cli.js worker --port 19100 --host 127.0.0.1 --host-id w-47 --instance-host 127.0.0.1 --log-level info
Restart=on-failure
RestartSec=3
KillMode=mixed
[Install]
WantedBy=multi-user.target
UNIT
systemctl daemon-reload; systemctl enable dshs-worker >/dev/null 2>&1
systemctl restart dshs-worker; sleep 5
echo " dshs-worker=$(systemctl is-active dshs-worker) healthz=$(curl -s -m 6 http://127.0.0.1:19100/healthz | head -c 120)"
echo "=== ④ 既有用户归属预置为 w-47(粘性锚点) ==="
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"insert into dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
select 'dsh-'||id, id, 'main', 'stopped', 'w-47', 0, 0, 0 from users
on conflict (id) do update set host_id='w-47', lease_until=0" 2>&1 | tail -1
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select u.username||' -> '||coalesce(i.host_id,'NULL') from users u left join dsh_instances i on i.user_id=u.id" 2>&1 | sed 's/^/ /'
echo "=== ⑤ 注册两台 worker ==="
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"insert into dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status)
values ('w-47','http://127.0.0.1:19100','$W47_TOKEN',1024,0,'up'),
('w-106','http://127.0.0.1:19000','$W106_TOKEN',2560,0,'up')
on conflict (id) do update set endpoint=excluded.endpoint, agent_token=excluded.agent_token, capacity_mb=excluded.capacity_mb, status='up'" 2>&1 | tail -1
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select id||' cap='||capacity_mb||' status='||status||' ep='||endpoint from dsh_hosts order by id" 2>&1 | sed 's/^/ /'
echo "=== 回滚剧本(现在就记下) ==="
echo " rm -f /etc/systemd/system/dshs.service.d/cluster.conf && systemctl daemon-reload && \\"
echo " systemctl restart dshs # 回 SQLite 单机;lib 回滚 = cp -a /opt/dsh/backups/lib-$TS/lib /opt/dshs/lib"
echo " 备份时间戳: $TS"
+20
View File
@@ -0,0 +1,20 @@
#!/usr/bin/env bash
# A 步:把 47 的生产库 SQLite → 本机控制面 PG(停机窗口内做,避免迁移期间库变动)
set -uo pipefail
PGURL="postgres://dshs:[email protected]:15432/dshs"
DB=/var/lib/dshs/dshs.db
cd /opt/dshs-cluster || exit 1
echo "=== 1) 停 dshs(窗口开始) ==="
systemctl stop dshs
sleep 2
echo " dshs=$(systemctl is-active dshs) 实例 scope 残留: $(systemctl list-units --type=scope --all 2>/dev/null | grep -c 'dsh-' || echo 0)"
echo " 库文件: $(ls -l $DB $DB-wal 2>/dev/null | awk '{print $5}' | tr '\n' '/')"
echo "=== 2) dry-run(只报行数) ==="
node scripts/migrate-sqlite-to-pg.mjs --sqlite "$DB" --pg "$PGURL" --dry-run 2>&1 | tail -18
echo "=== 3) 真迁 ==="
node scripts/migrate-sqlite-to-pg.mjs --sqlite "$DB" --pg "$PGURL" 2>&1 | tail -18
echo " rc=$?"
echo " ⏸ 窗口保持关闭 —— 部署 worker 与 drop-in 后再一起开(见后续步骤)"
+21
View File
@@ -0,0 +1,21 @@
#!/usr/bin/env bash
# 校验:SQLite 与 PG 两侧的 users 明细 + 与磁盘目录对照(判断 2 vs 7 是孤儿目录还是迁移漏行)
set -uo pipefail
cd /opt/dshs-cluster || exit 1
echo "=== SQLite 侧(只读打开,含 WAL) ==="
node -e '
const D = require("better-sqlite3");
const db = new D("/var/lib/dshs/dshs.db", { readonly: true });
const users = db.prepare("select id, username, role, uid from users order by rowid").all();
console.log(" users 行数:", users.length);
for (const u of users) console.log(` ${u.username} role=${u.role} uid=${u.uid} id=${u.id}`);
console.log(" credential_vault:", db.prepare("select count(*) c from credential_vault").get().c);
console.log(" business_plugins:", db.prepare("select count(*) c from business_plugins").get().c);
'
echo "=== PG 侧 ==="
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select username||' role='||role||' uid='||coalesce(uid::text,'NULL')||' id='||id from users order by rowid" 2>&1 | sed 's/^/ /'
echo " PG users 总数: $(PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc 'select count(*) from users')"
echo "=== 磁盘目录 vs DB ==="
echo " 目录($(ls /var/lib/dshs/users | wc -l) 个):"
ls /var/lib/dshs/users | sed 's/^/ /'
+22
View File
@@ -0,0 +1,22 @@
#!/usr/bin/env bash
# B1 步(在 47 上跑):uid 保真校验 + 建 47→106 专用密钥并打印公钥
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
echo "=== 1) uid 保真校验(PG 与 SQLite 必须一致 —— 否则 106 上文件属主全错) ==="
echo -n " PG : "; PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
"select string_agg(username||'='||coalesce(uid::text,'NULL'), ' ' order by row_id) from users" 2>&1 | head -1
echo -n " SQLite : "; node -e 'const D=require("better-sqlite3");const db=new D("/var/lib/dshs/dshs.db",{readonly:true});console.log(db.prepare("select username, uid from users order by rowid").all().map(u=>u.username+"="+u.uid).join(" "))'
echo " --- 磁盘目录属主(与 uid 对照;多出的 5 个是已删用户孤儿目录) ---"
for d in /var/lib/dshs/users/*/; do printf " %-38s uid=%s\n" "$(basename "$d")" "$(stat -c %u "$d")"; done
echo "=== 2) 建 47→106 专用密钥(仅用于 rsync) ==="
[ -f "$KEY" ] || ssh-keygen -t ed25519 -N "" -C "dshs-rsync-47to106" -f "$KEY" >/dev/null 2>&1
echo " PUBKEY=$(cat "$KEY.pub")"
echo "=== 3) 试连通 106:22 ==="
if ssh -i "$KEY" -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=8 [email protected] 'echo ok' 2>/dev/null | grep -q ok; then
echo " ✓ 已可连通(公钥已装)"
else
echo " ⏳ 尚不可连通 —— 需先把我本机把这个公钥装到 106"
fi
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env bash
# 切换 B 步(在 47 上跑):加 systemd drop-in → 重启 dshs(= 切换时刻,约 1-2 秒中断)
# · 用 drop-in 而非改 unit:unit 本体 hash 不变 ⇒ 回滚只需删 drop-in
# · 同时把 106 的 worker lib 也更新到同一版本(两侧代码必须一致)
set -uo pipefail
mkdir -p /etc/systemd/system/dshs.service.d
cat > /etc/systemd/system/dshs.service.d/cluster.conf <<CONF
# T08 集群化(2026-09-15):Manager 在 47、实例落在 w-47(既有用户)/ w-106(新用户)
# 回滚:删除本文件 → systemctl daemon-reload → systemctl restart dshs
[Service]
Environment="DSHS_DEPLOY_MODE=cluster"
Environment="DSHS_DB_URL=postgres://dshs:[email protected]:15432/dshs"
Environment="DSHS_CLUSTER_HOST_ID=w-47"
Environment="DSHS_CLUSTER_AGENT_URL=http://127.0.0.1:19100"
Environment="DSHS_CLUSTER_AGENT_TOKEN=dshs-worker-47-c4b7e19f"
Environment="DSHS_CLUSTER_INSTANCE_HOST=127.0.0.1"
Environment="DSHS_CLUSTER_WORKER_DATA_ROOT=/var/lib/dshs"
Environment="DSHS_CLUSTER_CAPACITY_MB=-1"
Environment="DSHS_CLUSTER_REGISTER_SELF=0"
Environment="DSHS_CLUSTER_LEASE_TTL_MS=30000"
CONF
echo " ✓ drop-in 已写($(wc -l < /etc/systemd/system/dshs.service.d/cluster.conf) 行)"
systemctl daemon-reload
systemctl restart dshs
sleep 6
echo "=== 切换后自检 ==="
echo " dshs=$(systemctl is-active dshs) | dshs-pg=$(systemctl is-active dshs-pg) | dshs-worker=$(systemctl is-active dshs-worker)"
echo " 生效 env(systemd 解析后):"
systemctl show dshs -p Environment 2>/dev/null | tr ' ' '\n' | grep -E "DEPLOY_MODE|CLUSTER_HOST_ID|CLUSTER_AGENT_URL|DB_URL" | sed 's/^/ /'
echo " 门户: login.html=$(curl -s -o /dev/null -w '%{http_code}' -m 8 http://127.0.0.1:3080/login.html)"
echo " 公网: https://alotbuy.com/login.html = $(curl -s -o /dev/null -w '%{http_code}' -m 12 https://alotbuy.com/login.html)"
echo " --- dshs cluster status ---"
cd /opt/dshs && set -a && . /etc/dshs.env && set +a && set -a && . /etc/systemd/system/dshs.service.d/cluster.conf 2>/dev/null || true
cd /opt/dshs && DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" node lib/cli.js cluster status 2>&1 | head -8 | sed 's/^/ /'
echo " --- 回滚命令(随时可用) ---"
echo " rm -f /etc/systemd/system/dshs.service.d/cluster.conf && systemctl daemon-reload && systemctl restart dshs"
+33
View File
@@ -0,0 +1,33 @@
#!/usr/bin/env bash
# B2 步(在 47 上跑):rsync 生产 dataRoot → 106
# · --numeric-ids 保 uid/gid(否则 106 上文件属主全错 ⇒ 实例 EACCES)
# · 排除 dshs.db*(DB 权威源已是 47 的 PG)与 secret.key(凭据主密钥不外扩到 Worker —— R5 最小面)
# · ⚠️ 所有 ssh 调用带 -n:脚本本身经 stdin 传入,ssh 若不隔离 stdin 会把**脚本剩余部分**吃掉
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
DST=[email protected]
SRC=/var/lib/dshs
SSHOPT="-n -i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=10"
command -v rsync >/dev/null || { dnf -y install rsync >/tmp/dnf-rsync.log 2>&1 && echo " ✓ 47 装上 rsync"; }
echo "=== 连通性 + 对端 rsync ==="
ssh $SSHOPT "$DST" 'command -v rsync >/dev/null || dnf -y install rsync >/tmp/dnf-rs.log 2>&1; echo " 对端: $(rsync --version | head -1)"' 2>&1 | tail -2
echo "=== 目标端准备 ==="
ssh $SSHOPT "$DST" 'mkdir -p /var/lib/dshs && ls -ld /var/lib/dshs' 2>&1 | sed 's/^/ /'
echo "=== rsync ==="
rsync -a --numeric-ids --stats -e "ssh $SSHOPT" \
--exclude 'dshs.db' --exclude 'dshs.db-shm' --exclude 'dshs.db-wal' --exclude 'secret.key' \
"$SRC/" "$DST:/var/lib/dshs/" 2>&1 | grep -E "Number of regular files transferred|Total file size|sent [0-9]|total size is" | sed 's/^/ /'
echo "=== 目标端核对 ==="
ssh $SSHOPT "$DST" 'bash -c "
echo \" 顶层: \$(ls /var/lib/dshs | tr \"\n\" \" \")\"
echo \" users 目录数: \$(ls /var/lib/dshs/users 2>/dev/null | wc -l)\"
for d in /var/lib/dshs/users/*/; do printf \" %-38s uid=%s\n\" \"\$(basename \$d)\" \"\$(stat -c %u \$d)\"; done
echo \" bundled-skills: \$(ls /var/lib/dshs/bundled-skills 2>/dev/null | wc -l) 项\"
echo \" business-plugins: \$(ls /var/lib/dshs/business-plugins 2>/dev/null | wc -l) 项\"
echo \" ⛔ 不应存在(dshs.db/secret.key): \$(ls /var/lib/dshs/dshs.db /var/lib/dshs/secret.key 2>/dev/null | wc -l) 个(应为 0)\"
"' 2>&1
+30
View File
@@ -0,0 +1,30 @@
#!/usr/bin/env bash
# B2' 步:47 → 106 直推生产 dataRoot(tar-over-ssh;只用命令执行,绕开 rsync 协议问题)
# · tar --numeric-owner 保 uid/gid(否则 106 上属主错 ⇒ 实例 EACCES)
# · 排除 dshs.db*(权威源=47 的 PG)与 secret.key(主密钥不外扩 —— R5 最小面)
set -uo pipefail
KEY=/root/.ssh/dshworker_ed25519
DST=[email protected]
SSHO="-i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new"
echo "=== 1) 目标端准备(ssh -n 防抢 stdin) ==="
ssh -n $SSHO "$DST" 'mkdir -p /var/lib/dshs && echo " ready: $(ls -ld /var/lib/dshs)"'
echo "=== 2) 推送(tar → ssh → tar --numeric-owner -x) ==="
cd /var/lib/dshs
tar --numeric-owner -cf - \
--exclude=./dshs.db --exclude=./dshs.db-shm --exclude=./dshs.db-wal --exclude=./secret.key \
. | ssh $SSHO "$DST" 'tar -C /var/lib/dshs --numeric-owner -xf -'
rc=$?
echo " 管道 rc=$rc"
echo "=== 3) 目标端核对 ==="
ssh -n $SSHO "$DST" 'bash -c "
echo \" 顶层: \$(ls /var/lib/dshs | tr \"\n\" \" \")\"
echo \" 总量: \$(du -sh /var/lib/dshs | cut -f1)\"
echo \" users 目录数: \$(ls /var/lib/dshs/users | wc -l)\"
echo \" --- 属主抽样(应与 47 的 uid 一致) ---\"
for d in /var/lib/dshs/users/*/; do printf \" %-38s uid=%s\n\" \"\$(basename \$d)\" \"\$(stat -c %u \$d)\"; done
echo \" bundled-skills=\$(ls /var/lib/dshs/bundled-skills | wc -l) business-plugins=\$(ls /var/lib/dshs/business-plugins | wc -l)\"
echo \" ⛔ 不该有 dshs.db/secret.key: \$(ls /var/lib/dshs/dshs.db /var/lib/dshs/secret.key 2>/dev/null | wc -l) 个(应 0)\"
"'
+79
View File
@@ -0,0 +1,79 @@
#!/usr/bin/env bash
# 最终验证(在 47 上跑):清污染 → 重置锚点 → 重验两条路径
# ① 既有用户 guest:留 w-47 + 工作区有历史数据 + 实例页正常
# ② 新用户:落 w-106 + **文件真的写到 106 的盘** + 实例页正常
set -uo pipefail
export PGPASSWORD=dshs_cluster_2026
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
M=http://127.0.0.1:3080
DOMAIN=alotbuy.com
T47=dshs-worker-47-c4b7e19f
T106=dshs-worker-7f3a91c05e
SSH106="ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected]"
mksess() {
local u="$1" T H
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
Q "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+1800000, '127.0.0.1','switch-verify' from users where username='$u'" >/dev/null
printf '%s' "$T"
}
owner() { Q "select coalesce(i.host_id,'NULL')||' epoch='||coalesce(i.epoch,0) from users u left join dsh_instances i on i.user_id=u.id where u.username='$1'"; }
agent() { curl -s -m 6 -H "x-dsh-agent-token: $2" "http://127.0.0.1:$1/healthz" | grep -o '"instances":[0-9]*'; }
isapp() { grep -q '<base href=' <<<"$1" && echo "✓实例页" || echo "✗非实例页"; }
echo "########## 0) 清污染:停所有实例 + 删测试用户 ##########"
systemctl restart dshs-worker; sleep 4
$SSH106 'systemctl restart dshs-worker' >/dev/null 2>&1; sleep 4
echo " 重启后 w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
AT=$(mksess admin); AC="sid=$AT"
for u in $(Q "select id from users where username like 'switchtest%' or username like 'swtest%'"); do
echo " 删测试用户 $u → $(curl -s -m 20 -X DELETE -b "$AC" "$M/api/admin/users/$u" -o /dev/null -w '%{http_code}')"
done
echo " 剩余用户: $(Q "select string_agg(username,', ') from users")"
echo "########## 1) 重置 guest 锚点(host_id=w-47) ##########"
Q "update dsh_instances set host_id='w-47', epoch=0, lease_until=0, status='stopped'
where user_id=(select id from users where username='guest')" >/dev/null
echo " $(owner guest)"
echo
echo "########## 2) 路径①:既有用户 guest ##########"
GT=$(mksess guest); GC="sid=$GT"
E=$(curl -s -m 60 -X POST -b "$GC" -H 'content-type: application/json' -d '{}' "$M/api/dsh/enter")
sleep 3
echo " 归属: $(owner guest) ← 期望 w-47"
echo " w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
echo " 工作区条目: $(curl -s -m 10 -b "$GC" "$M/api/desktop/tree" | grep -o '"name":"[^"]*"' | head -4 | tr '\n' ' ')"
PAGE=$(curl -s -m 25 -L -b "$GC" -H "Host: guest.$DOMAIN" "$M/" | head -c 300)
echo " 实例页: $(isapp "$PAGE")"
echo
echo "########## 3) 路径②:新用户 ##########"
NU="swtest2$(date +%H%M%S)"
echo " register=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" "$M/api/auth/register" -o /dev/null -w '%{http_code}')"
NID=$(Q "select id from users where username='$NU'")
echo " approve=$(curl -s -m 15 -X POST -b "$AC" "$M/api/admin/users/$NID/approve" -o /dev/null -w '%{http_code}')"
NC=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" -D - "$M/api/auth/login" -o /dev/null | grep -i '^set-cookie' | head -1 | grep -oP 'sid=[^;]+')
echo " mkdir=$(curl -s -m 15 -X POST -b "$NC" -H 'content-type: application/json' -d '{"path":"proj"}' "$M/api/fs/mkdir" -o /dev/null -w '%{http_code}') ← 首次触达应把归属钉住"
echo " 钉住后归属: $(owner "$NU") ← 期望 w-106(与下面的 launch 必须同台)"
echo " upload=$(curl -s -m 20 -X POST -b "$NC" -H 'content-type: application/json' -d "{\"path\":\"proj\",\"name\":\"hello.txt\",\"data\":\"$(printf 'hi-from-switch' | base64 -w0)\"}" "$M/api/fs/upload" -o /dev/null -w '%{http_code}')"
echo " launch=$(curl -s -m 60 -X POST -b "$NC" -H 'content-type: application/json' -d '{"folder":"proj"}' "$M/api/dsh/launch" -o /dev/null -w '%{http_code}')"
sleep 4
echo " 归属: $(owner "$NU") ← 期望 w-106(粘性保持)"
echo " w-106=$(agent 19000 $T106) w-47=$(agent 19100 $T47)"
echo " --- 落盘取证 ---"
L47=$($SSH106 "ls /var/lib/dshs/users/$NID/ws/proj/hello.txt 2>/dev/null" 2>/dev/null || true)
echo " 106 盘: ${L47:-不存在}"
echo " 47 盘: $(ls /var/lib/dshs/users/$NID/ws/proj/hello.txt 2>/dev/null || echo 不存在(应不存在 ✓))"
NP=$(curl -s -m 25 -L -b "$NC" -H "Host: $NU.$DOMAIN" "$M/" | head -c 300)
echo " 实例页(Host: $NU.$DOMAIN): $(isapp "$NP")"
echo
echo "########## 收尾 ##########"
curl -s -m 40 -X POST -b "$GC" "$M/api/dsh/stop" -o /dev/null -w " guest stop=%{http_code}\n"
curl -s -m 40 -X POST -b "$NC" "$M/api/dsh/stop" -o /dev/null -w " newuser stop=%{http_code}\n"
echo " stop 后 guest 归属(**不应再被清空**): $(owner guest)"
Q "delete from sessions where user_agent='switch-verify'" >/dev/null
echo " 残留临时 session: $(Q "select count(*) from sessions where user_agent='switch-verify'")(应 0)"
echo " 新用户待清: $NU / $NID"
+63
View File
@@ -0,0 +1,63 @@
#!/usr/bin/env bash
# 切换后功能验证 v2(在 47 上跑)
# ① 既有用户 guest:应留 w-47,且**工作区有历史数据**(文件在 47 的盘上)
# ② 新用户:应落 w-106,且**建的文件真的出现在 106 的盘上**(文件面路由已修)
# 判据:实例页必须带 <base href="/"(门户页不算);归属看 PG;文件落盘看两台机器磁盘
set -uo pipefail
PG() { PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
M=http://127.0.0.1:3080
DOMAIN=alotbuy.com
T47=dshs-worker-47-c4b7e19f
T106=dshs-worker-7f3a91c05e
mksess() {
local u="$1" T H
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
PG "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+1800000, '127.0.0.1','switch-verify' from users where username='$u'" >/dev/null
printf '%s' "$T"
}
owner() { PG "select u.username||' -> '||coalesce(i.host_id,'NULL')||' epoch='||i.epoch from users u left join dsh_instances i on i.user_id=u.id where u.username='$1'"; }
agent() { curl -s -m 6 -H "x-dsh-agent-token: $2" "http://127.0.0.1:$1/healthz" | grep -o '"instances":[0-9]*'; }
isapp() { grep -q '<base href=' <<<"$1" && echo "✓实例页" || echo "✗非实例页"; }
echo "############ ① 既有用户 guest(应留 w-47 + 工作区有数据) ############"
GT=$(mksess guest); GC="sid=$GT"
E=$(curl -s -m 60 -X POST -b "$GC" -H 'content-type: application/json' -d '{}' "$M/api/dsh/enter")
echo " enter → $(head -c 130 <<<"$E")"
sleep 3
echo " 归属: $(owner guest) ← 期望 w-47"
echo " w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
echo " 工作区首项: $(curl -s -m 10 -b "$GC" "$M/api/desktop/tree" | head -c 150)"
URL=$(grep -o '"url":"[^"]*"' <<<"$E" | head -1 | cut -d'"' -f4)
PAGE=$(curl -s -m 25 -L -b "$GC" -H "Host: guest.$DOMAIN" "$M/" | head -c 300)
echo " 实例页: $(isapp "$PAGE")"
echo " 47 盘上 guest 工作区条目: $(ls /var/lib/dshs/users/4092b965-2f68-4977-9989-68b3966f7df0/ws 2>/dev/null | wc -l) 项(>0 = 数据在 47 ✓)"
echo
echo "############ ② 新用户(应落 w-106 + 文件真的写到 106 盘) ############"
NU="swtest$(date +%H%M%S)"
AT=$(mksess admin); AC="sid=$AT"
echo " register=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" "$M/api/auth/register" -o /dev/null -w '%{http_code}')"
NU_ID=$(PG "select id from users where username='$NU'")
echo " approve=$(curl -s -m 15 -X POST -b "$AC" "$M/api/admin/users/$NU_ID/approve" -o /dev/null -w '%{http_code}')"
NC=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" -D - "$M/api/auth/login" -o /dev/null | grep -i '^set-cookie' | head -1 | grep -oP 'sid=[^;]+')
echo " mkdir=$(curl -s -m 15 -X POST -b "$NC" -H 'content-type: application/json' -d '{"path":"proj"}' "$M/api/fs/mkdir" -o /dev/null -w '%{http_code}')"
echo " upload=$(curl -s -m 20 -X POST -b "$NC" -H 'content-type: application/json' -d "{\"path\":\"proj\",\"name\":\"hello.txt\",\"data\":\"$(printf 'hi-from-switch' | base64 -w0)\"}" "$M/api/fs/upload" -o /dev/null -w '%{http_code}')"
echo " launch=$(curl -s -m 60 -X POST -b "$NC" -H 'content-type: application/json' -d '{"folder":"proj"}' "$M/api/dsh/launch" -o /dev/null -w '%{http_code}')"
sleep 4
echo " 归属: $(owner "$NU") ← 期望 w-106"
echo " w-106=$(agent 19000 $T106) w-47=$(agent 19100 $T47)"
echo " --- 文件落盘取证(这才是文件面路由修好的证据) ---"
echo " 47 盘: $(ls /var/lib/dshs/users/$NU_ID/ws/proj/hello.txt 2>/dev/null && echo 存在 || echo '不存在 ✓(不应在 47)')"
echo " 106 盘: $(ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] "ls /var/lib/dshs/users/$NU_ID/ws/proj/hello.txt 2>/dev/null" 2>/dev/null && echo 存在✓ || echo '不存在 ✗')"
NU_PAGE=$(curl -s -m 25 -L -b "$NC" -H "Host: $NU.$DOMAIN" "$M/" | head -c 300)
echo " 实例页(Host: $NU.$DOMAIN): $(isapp "$NU_PAGE")"
echo
echo "############ 收尾 ############"
curl -s -m 40 -X POST -b "$GC" "$M/api/dsh/stop" -o /dev/null -w " guest stop=%{http_code}\n"
curl -s -m 40 -X POST -b "$NC" "$M/api/dsh/stop" -o /dev/null -w " newuser stop=%{http_code}\n"
PG "delete from sessions where user_agent='switch-verify'" >/dev/null
echo " 残留临时 session: $(PG "select count(*) from sessions where user_agent='switch-verify'")(应 0)"
echo " 新用户记录: $NU / $NU_ID"
+53
View File
@@ -0,0 +1,53 @@
#!/usr/bin/env bash
# C 步(在 106 上跑):把生产 Worker agent 装成 systemd 单元
# · env 与 47 的生产实例侧对齐(DSH_INSTANCE_* 必须一致 —— 实例是在 Worker 上 spawn 的)
# · token 走 env 文件(600)而不是命令行,避免 ps 泄露
# · 反向隧道复用演练时那把 key(其公钥已在 47 的 authorized_keys 里,restrict,port-forwarding)
set -uo pipefail
TOKEN="${WORKER_TOKEN:-dshs-worker-7f3a91c05e}"
cat > /etc/dshs-worker.env <<ENV
# DSHS 集群 Worker(2026-09-15 切换)—— 与 47 /etc/dshs.env 的**实例侧**条目保持一致
DSHS_DATA_ROOT=/var/lib/dshs
DSHS_ISOLATION_MODE=account
DSHS_DSH_BIN=/usr/bin/dsh
DSHS_BASE_UID=100000
DSH_INSTANCE_NODE_OPTIONS=--max-old-space-size=160
DSH_INSTANCE_UNIVER_SOCKET=auto
# 控制通道:Worker 主动拨 47 的反向隧道(公网入方向被云安全组挡住 ⇒ 只能这个方向)
DSHS_CLUSTER_AGENT_TOKEN=$TOKEN
[email protected]:32022
DSHS_TUNNEL_IDENTITY=/root/.ssh/tunnel_ed25519
ENV
chmod 600 /etc/dshs-worker.env
cat > /etc/systemd/system/dshs-worker.service <<UNIT
[Unit]
Description=DSHS cluster worker agent (hosts per-user dsh instances on 106)
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
EnvironmentFile=/etc/dshs-worker.env
ExecStart=/usr/bin/node /opt/dshs-cluster/lib/cli.js worker --port 19000 --host 127.0.0.1 --host-id w-106 --instance-host 127.0.0.1 --log-level info
Restart=on-failure
RestartSec=3
KillMode=mixed
[Install]
WantedBy=multi-user.target
UNIT
systemctl daemon-reload
systemctl enable dshs-worker >/dev/null 2>&1
echo " 单元已装并 enable;token 长度=${#TOKEN}"
# 旧的手工 agent 若在跑先停(按端口定位)
pid=$(ss -lntpH 'sport = :19000' 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
[ -n "$pid" ] && kill "$pid" && sleep 2 && echo " 已停旧的手工 agent pid=$pid"
systemctl restart dshs-worker
sleep 6
echo " dshs-worker=$(systemctl is-active dshs-worker)"
echo " healthz: $(curl -s -m 6 http://127.0.0.1:19000/healthz | head -c 200)"
+36
View File
@@ -0,0 +1,36 @@
#!/usr/bin/env bash
# 切换收尾:清理验证残留 + 健康检查(在 47 上跑)
set -uo pipefail
export PGPASSWORD=dshs_cluster_2026
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
M=http://127.0.0.1:3080
echo "=== 1) 删掉验证留下的测试用户 ==="
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
Q "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+600000, '127.0.0.1','switch-cleanup' from users where username='admin'" >/dev/null
for u in $(Q "select id from users where username like 'swtest%' or username like 'switchtest%'"); do
echo " delete $u → $(curl -s -m 20 -X DELETE -b "sid=$T" "$M/api/admin/users/$u" -o /dev/null -w '%{http_code}')"
done
Q "delete from sessions where user_agent='switch-cleanup'" >/dev/null
echo " 剩余用户: $(Q "select string_agg(username||'('||role||')', ', ') from users")"
echo "=== 2) 停掉验证期间起的实例 ==="
systemctl restart dshs-worker; sleep 4
ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] 'systemctl restart dshs-worker' >/dev/null 2>&1
sleep 4
echo " w-47=$(curl -s -m 6 -H 'x-dsh-agent-token: dshs-worker-47-c4b7e19f' http://127.0.0.1:19100/healthz | grep -o '\"instances\":[0-9]*')"
echo " w-106=$(ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] 'curl -s -m 6 -H "x-dsh-agent-token: dshs-worker-7f3a91c05e" http://127.0.0.1:19000/healthz | grep -o .instances.:[0-9]*' 2>/dev/null)"
echo "=== 3) 健康检查 ==="
echo " dshs=$(systemctl is-active dshs) dshs-pg=$(systemctl is-active dshs-pg) dshs-worker=$(systemctl is-active dshs-worker)"
echo " 门户公网: $(curl -s -o /dev/null -w '%{http_code}' -m 12 https://alotbuy.com/login.html)"
echo " --- dshs cluster status ---"
cd /opt/dshs && set -a && . /etc/dshs.env && set +a
DSHS_DEPLOY_MODE=cluster DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" \
node lib/cli.js cluster status 2>&1 | head -9 | sed 's/^/ /'
echo " --- dshs doctor ---"
DSHS_DEPLOY_MODE=cluster DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" \
node lib/cli.js doctor 2>&1 | grep -cE "^✓" | sed 's/^/ ✓ 项数: /'
echo " --- 实例归属总览 ---"
Q "select u.username || ' → ' || coalesce(i.host_id,'(未指派)') || ' status=' || coalesce(i.status,'-') from users u left join dsh_instances i on i.user_id=u.id order by u.row_id" | sed 's/^/ /'
+31
View File
@@ -0,0 +1,31 @@
#!/usr/bin/env bash
# 拆除演练环境(为生产切换让路):
# · 47:演练 Manager(127.0.0.1:13080)
# · 106:两个演练 agent(19000/19001)⇒ 会 teardown 实例并关闭隧道
# 目的:① 释放 47 的 15432(隧道转发占用 → 生产 PG 要用)
# ② 避免"演练 Manager + 生产 Manager 抢同一个 agent"
set -uo pipefail
echo "=== 47 侧:停演练 Manager(按端口定位,绝不碰生产 3080) ==="
PID=$(ss -lntpH 'sport = :13080' 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
if [ -n "$PID" ]; then kill "$PID" && echo " 已停演练 Manager pid=$PID"; else echo " 13080 无监听"; fi
sleep 2
echo "=== 106 侧:停两个演练 agent ==="
for port in 19000 19001; do
pid=$(ss -lntpH "sport = :$port" 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
[ -n "$pid" ] && kill "$pid" && echo " 已停 agent($port) pid=$pid" || echo " $port 无监听"
done
sleep 4
echo " 残留 dsh 实例: $(pgrep -cf 'dsh --profile' || echo 0)"
echo " 隧道进程: $(pgrep -cf 'tunnel_ed25519' || echo 0)"
echo "=== 47 侧:15432 是否已释放(隧道转发应已消失) ==="
ss -lntp 2>/dev/null | grep 15432 || echo " ✓ 15432 已空闲"
echo "=== 加固:若隧道 sshd 残留,按端口收掉 ==="
for port in 19000 19001 15432; do
pid=$(ss -lntpH "sport = :$port" 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
[ -n "$pid" ] && kill "$pid" 2>/dev/null && echo " 收掉残留监听 $port pid=$pid"
done
sleep 2
ss -lntp 2>/dev/null | grep -E "15432|1900[01]" || echo " ✓ 三个端口都已空闲"
+195
View File
@@ -0,0 +1,195 @@
/**
* T08 S3 · 端到端验证:**Manager 经 RemoteSpawner 把实例起在 worker agent 上**。
*
* 与 `smoke-dsh.mjs` 的区别:那条走的是"本机直接 spawn",这条**多了一跳 HTTP**
* (Manager → agent → LocalSpawner),因此它验证的是 S3 真正的交付物:
* ① 路由/代理层**一行没改**就能工作(`Spawner` 抽象 + `endpointFor` 的 host:port);
* ② **launch token 回传**(P0-6)—— 否则"登录直达会话"与 401 自愈会失效;
* ③ **幂等键**:同一 operationId 重发不会起第二个实例(Manager 超时重试是常态);
* ④ **self-fencing**:`/fence` 下发的 epoch 更高时,agent 主动停掉自己那个实例。
*
* 刻意用 **soft 隔离 + stand-in fake-dsh**:本测试要验的是**跨机协议**,
* 不是沙箱(沙箱另有 S1.6 的双机证据)。用 account 模式反而会被"夹具路径必须在
* 沙箱绑定集内"这条夹具限制干扰(见 `交接单/T08-§10.4`)。
*
* 运行:node scripts/verify-cluster-agent.mjs
*/
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
const TOKEN = 'verify-cluster-agent-token'
const dataRoot = mkdtempSync(join(tmpdir(), 'dsh-cluster-'))
let agentApp
let agentHandle
let app
try {
// ── 1) 起 worker agent(进程内,端口随机)──────────────────────────────
const agentConfig = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot,
dshCommand: [process.execPath, fakeDsh],
clusterHostId: 'w-1',
})
const agent = buildWorkerAgent(agentConfig, {
hostId: 'w-1',
token: TOKEN,
port: 0,
host: '127.0.0.1',
instanceHost: '127.0.0.1',
logLevel: 'warn',
})
agentApp = agent.app
agentHandle = agent // 收尾要用 agent.stop()(会 teardown 本机实例),否则子进程孤儿化
await agentApp.listen({ host: '127.0.0.1', port: 0 })
const agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
console.log('agent ->', agentUrl)
const agentJson = async (path, { method = 'GET', body } = {}) => {
const res = await fetch(agentUrl + path, {
method,
headers: {
[AGENT_TOKEN_HEADER]: TOKEN,
...(body ? { 'content-type': 'application/json' } : {}),
},
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null }
}
// agent 存活 + 鉴权(不带 token 必须 401)
const hz = await agentJson('/healthz')
assert(hz.status === 200 && hz.body.hostId === 'w-1', 'agent healthz')
const noAuth = await fetch(agentUrl + '/instances')
assert(noAuth.status === 401, 'agent 拒绝无凭据请求')
// ── 2) 起 Manager(deployMode=cluster → RemoteSpawner)─────────────────
const managerConfig = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot,
deployMode: 'cluster',
clusterAgentUrl: agentUrl,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
})
app = await buildServer(managerConfig)
await app.listen({ port: 0 })
const base = `http://127.0.0.1:${app.server.address().port}`
console.log('manager ->', base, '(deployMode=cluster)')
await app.db.createUser({
id: 'u1',
username: 'carol',
passHash: await hashPassword('carolpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
mkdirSync(join(dataRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: {
...(body ? { 'content-type': 'application/json' } : {}),
...(cookie ? { cookie } : {}),
},
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
}
// ── 3) 登录 → 拉起(实例实际落在 agent 上)────────────────────────────
let r = await json('/api/auth/login', { method: 'POST', body: { username: 'carol', password: 'carolpass123' } })
assert(r.status === 200, 'login succeeds')
const cookie = r.setCookie.split(';')[0]
r = await json('/api/dsh/status', { cookie })
assert(r.body.running === false, 'not running initially')
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
console.log('launch ->', r.status, r.body?.url ? 'url 已返回' : r.body)
assert(r.status === 200, 'launch succeeds')
// ② launch token 回传(P0-6):URL 里必须带 token,否则"登录直达"失效
assert(typeof r.body.url === 'string' && r.body.url.includes('token='), 'launch token 必须回传到 URL')
// 实例真的在 **worker** 上(而不是 Manager 本机)
const onAgent = await agentJson('/instances')
assert(onAgent.body.instances.length === 1, 'worker 上有 1 个实例')
assert(onAgent.body.instances[0].userId === 'u1', 'worker 上的实例属于 u1')
console.log('agent 视角 -> 实例数', onAgent.body.instances.length)
r = await json('/api/dsh/status', { cookie })
assert(r.body.running === true, 'running after launch')
// ── 4) 代理链路(endpointFor → agent 给的 host:port)──────────────────
let proxyText
for (let i = 0; i < 20; i += 1) {
try {
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie } })
if (res.status === 200) {
proxyText = await res.text()
break
}
} catch {
/* 子进程还没监听,重试 */
}
await new Promise((resolve) => setTimeout(resolve, 100))
}
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), 'proxy reaches the child DSH(经远端协议)')
console.log('proxy -> 200 且命中 fake-dsh')
// ── 5) 幂等键:同 operationId 重发不得起第二个实例 ─────────────────────
const opId = 'verify-idempotent-1'
const l1 = await agentJson('/launch', { method: 'POST', body: { userId: 'u1', folder: join(dataRoot, 'users', 'u1', 'ws', 'proj'), patch: undefined, operationId: opId } })
const l2 = await agentJson('/launch', { method: 'POST', body: { userId: 'u1', folder: join(dataRoot, 'users', 'u1', 'ws', 'proj'), patch: undefined, operationId: opId } })
assert(l1.status === 200 && l2.status === 200, '重复 launch 不报错')
const afterIdem = await agentJson('/instances')
assert(afterIdem.body.instances.length === 1, '幂等:仍然只有 1 个实例')
console.log('幂等 -> 同 operationId 重发后实例数仍为', afterIdem.body.instances.length)
// ── 6) self-fencing:更高 epoch 下发 ⇒ agent 主动停掉自己那个实例 ───────
await agentJson('/launch', { method: 'POST', body: { userId: 'u1', epoch: 1, operationId: 'verify-epoch-1' } })
const f1 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 1 } })
assert(f1.body.fenced === false, 'epoch 相同 ⇒ 不被 fence')
const f2 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 2 } })
assert(f2.body.fenced === true, 'epoch 更高 ⇒ self-fence')
const afterFence = await agentJson('/instances')
assert(afterFence.body.instances.length === 0, 'fence 后实例已停')
console.log('self-fence -> epoch 1→2 触发,实例已停止')
// ── 7) 停止 ───────────────────────────────────────────────────────────
r = await json('/api/dsh/stop', { method: 'POST', cookie })
assert(r.status === 200, 'stop succeeds')
r = await json('/api/dsh/status', { cookie })
assert(r.body.running === false, 'stopped after stop')
console.log('\nOK: cluster 模式(Manager → worker agent → 实例)端到端通过')
console.log(' ✓ 路由/代理层零改动 ✓ launch token 回传 ✓ 幂等键 ✓ self-fencing')
} finally {
await app?.close()
// ⚠️ 必须走 agent.stop():它先 teardown 本机实例再关 HTTP —— 否则 fake-dsh 孤儿会继承
// stdout,管道不关 ⇒ ssh / CI 挂死(2026-09-15 实测)。
await agentHandle?.stop()
await new Promise((resolve) => setTimeout(resolve, 500))
try {
rmSync(dataRoot, { recursive: true, force: true })
} catch {
// best-effort:Windows 上子进程的 cwd 还在里面时会 EBUSY(temp 目录会被系统回收)
}
}
+184
View File
@@ -0,0 +1,184 @@
/**
* T08 · **真跨机演练**驱动脚本(在 Manager 那台机器上运行)。
*
* 与 `verify-cluster-live.mjs`(同机、脚本自己起进程)的区别:这里**假设两侧都已部署好**:
* · Manager 运行在**本机**(47)`http://127.0.0.1:13080`
* · Worker agent 运行在**另一台机器**(106),经 **SSH 反向隧道**出现在本机 `127.0.0.1:19000`
* · 控制面 PG 也在**另一台机器**(106)上,经隧道出现在本机 `127.0.0.1:15432`
* 它回答的是本次演练的核心问题:**跨机到底能不能用**(含跨机代理取页面、跨 worker 迁移)。
*
* 运行(在 47 上):MANAGER=http://127.0.0.1:13080 AGENT_TOKEN=cross-machine-token \
* AGENT2=http://127.0.0.1:19001 node scripts/verify-cluster-cross.mjs
*/
const MANAGER = process.env.MANAGER ?? 'http://127.0.0.1:13080'
const TOKEN = process.env.AGENT_TOKEN ?? 'cross-machine-token'
const AGENT1 = process.env.AGENT1 ?? 'http://127.0.0.1:19000'
const AGENT2 = process.env.AGENT2 ?? ''
const ADMIN_PW = process.env.ADMIN_PW ?? 'crossmgr123'
const USER_PW = 'crossuser123'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(MANAGER + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
signal: AbortSignal.timeout(30_000),
})
const text = await res.text()
let parsed = null
try {
parsed = text === '' ? null : JSON.parse(text)
} catch {
parsed = { raw: text.slice(0, 200) }
}
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
}
/** 直接问 worker(绕过 Manager)—— 证明实例真的落在**那台机器**上。 */
const agent = async (base, path) => {
const res = await fetch(base + path, { headers: { 'x-dsh-agent-token': TOKEN }, signal: AbortSignal.timeout(10_000) })
const text = await res.text()
return { status: res.status, body: text === '' ? null : JSON.parse(text) }
}
/** 取页面:跟随重定向(dsh 首页 303),并对启动窗口的断连做重试。 */
async function fetchPage(url, cookie, tries = 40) {
let status = 0
let snippet = ''
for (let i = 0; i < tries; i += 1) {
try {
const res = await fetch(MANAGER + url, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(20_000) })
status = res.status
if (res.status === 200) {
snippet = (await res.text()).slice(0, 200)
break
}
} catch {
status = 0
}
await sleep(1000)
}
return { status, snippet }
}
async function waitRunning(cookie, tries = 60) {
for (let i = 0; i < tries; i += 1) {
const st = await json('/api/dsh/status', { cookie })
if (st.body?.running === true) return st.body
if (st.body?.instance?.status === 'crashed') return st.body
await sleep(1000)
}
return await json('/api/dsh/status', { cookie }).then((r) => r.body)
}
try {
console.log('=== 跨机演练:Manager=%s Worker=%s ===', MANAGER, AGENT1)
// ── 0) 两侧可达性(跨机链路的第一层证据)─────────────────────────────
const h1 = await agent(AGENT1, '/healthz')
assert(h1.status === 200 && h1.body.hostId === 'w-106', `Worker w-106 应可达(实际 ${JSON.stringify(h1.body)})`)
assert(h1.body.tunnel?.ready === true, `Worker 侧隧道应就绪(实际 ${JSON.stringify(h1.body.tunnel)})`)
console.log('⓪ worker 可达 -> %s(隧道 ready,已转发 %s)', h1.body.hostId, JSON.stringify(h1.body.tunnel.ports))
// ── 1) 管理面:注册 worker(join 脚本干的事)──────────────────────────
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
assert(adm.status === 200, `管理员登录失败 ${adm.status}`)
const adminCookie = adm.setCookie.split(';')[0]
let r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-106', endpoint: AGENT1, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-106 失败 ${r.status}`)
const hosts = await json('/api/admin/hosts', { cookie: adminCookie })
assert(hosts.body.hosts.some((h) => h.id === 'w-106'), 'w-106 出现在 worker 目录')
assert(!('agentToken' in (hosts.body.hosts[0] ?? {})), '**绝不下发 agentToken**')
console.log('① 注册 -> w-106(列表不含 agentToken)')
// ── 2) 用户流程 ───────────────────────────────────────────────────────
const uname = `crossuser${Date.now() % 100000}`
r = await json('/api/auth/register', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(r.status === 201, `注册应 201(实际 ${r.status})`)
const users = await json('/api/admin/users', { cookie: adminCookie })
const target = users.body.users.find((u) => u.username === uname)
assert(target !== undefined, '管理员能看到待审用户')
r = await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
assert(r.status === 200, `审批应 200(实际 ${r.status})`)
const login = await json('/api/auth/login', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(login.status === 200, `用户登录失败 ${login.status}`)
const cookie = login.setCookie.split(';')[0]
console.log('② 用户流程 -> 注册→审批→登录(uid=%s)', target.id)
// ── 3) 文件面跨机(Manager 在 47、目录落在 106)───────────────────────
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
console.log('③ 文件面 -> mkdir 经隧道落到 106 的 worker')
// ── 4) 拉起实例(真 dsh 在 **106** 上)────────────────────────────────
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
const st = await waitRunning(cookie)
assert(st?.running === true, `实例应 running(实际 ${JSON.stringify(st)?.slice(0, 300)})`)
const onAgent = await agent(AGENT1, '/instances')
assert(onAgent.body.instances.length === 1, 'worker(106) 上有 1 个实例')
console.log('④ 拉起 -> running=true,**实例在 106 上**(worker /instances=%d)', onAgent.body.instances.length)
// ── 5) 登录直达 + **跨机取页面**(本演练的核心证据)───────────────────
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter.status === 200, `enter 应 200(实际 ${enter.status})`)
const url = enter.body.url
assert(typeof url === 'string' && url.includes('token='), `enter 应带 token(实际 ${url})`)
const page = await fetchPage(url, cookie)
assert(page.status === 200, `**跨机页面**应 200(实际 ${page.status})`)
console.log('⑤ 跨机页面 -> 200(47 的 Manager 代理到 106 的实例;片段 %s)', page.snippet.replace(/\s+/g, ' ').slice(0, 60))
// ── 6) 第二台 worker(106 上模拟的第二台服务器)+ 迁移 ────────────────
if (AGENT2 !== '') {
const h2 = await agent(AGENT2, '/healthz')
assert(h2.status === 200, `第二台 worker 应可达(实际 ${h2.status})`)
r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-106b', endpoint: AGENT2, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-106b 失败 ${r.status}`)
console.log('⑥ 第二台 -> %s(模拟的第二台服务器)已注册', h2.body.hostId)
r = await json(`/api/admin/users/${target.id}/dsh/migrate`, {
method: 'POST',
cookie: adminCookie,
body: { targetHost: 'w-106b' },
})
assert(r.status === 200, `迁移应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(r.body.to === 'w-106b', `迁移目标应为 w-106b(实际 ${r.body.to})`)
await waitRunning(cookie)
const a1 = await agent(AGENT1, '/instances')
const a2 = await agent(AGENT2, '/instances')
assert(a1.body.instances.length === 0 && a2.body.instances.length === 1, '实例应从 w-106 移到 w-106b')
console.log('⑦ 跨机迁移 -> %s → %s(epoch=%d),源机已空、目标机有 1 个实例', r.body.from, r.body.to, r.body.epoch)
const enter2 = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter2.status === 200 && enter2.body.url !== url, '迁移后 enter 应给**新** URL')
const page2 = await fetchPage(enter2.body.url, cookie)
assert(page2.status === 200, `迁移后页面应 200(实际 ${page2.status})`)
console.log('⑧ 迁移后 -> 新 token URL 页面 200')
} else {
console.log('⑥⑦⑧ 跳过(未提供 AGENT2)')
}
// ── 9) 收尾 ───────────────────────────────────────────────────────────
r = await json('/api/dsh/stop', { method: 'POST', cookie })
assert(r.status === 200, `stop 应 200(实际 ${r.status})`)
console.log('⑨ 停止 -> ok')
console.log('\nOK: **真跨机**(47 当 Manager / 106 当 Worker,隧道跨界)演练通过')
console.log(' ✓ worker 可达 ✓ 注册 ✓ 用户流程 ✓ 文件面跨机 ✓ 实例在 106 ✓ 跨机取页面 ✓ 跨 worker 迁移')
} finally {
/* 不主动清理:实例由调用方决定留或停(演练后要观察现场) */
}
+167
View File
@@ -0,0 +1,167 @@
/**
* T08 · **域名形态访问**验证(在演练环境做:不动生产、不动 DNS、不动证书)。
*
* 要回答的问题:生产切到 cluster(Manager 在 47、实例在 106)后,
* **按域名形态访问**(`<用户名>.alotbuy.com`)还能不能正常落到 106 上的实例?
*
* 做法:给演练 Manager 设一个**测试 baseDomain**,用**显式 `Host` 头**打进去。
*
* ⚠️ 关键坑(2026-09-15 实际踩到,两次假阳性都源于它):**`fetch` 会静默丢弃 `Host` 头**
* (Fetch 规范把它列为禁止头,undici 直接忽略)⇒ 请求落到"无租户"的门户路由、回 200 门户页,
* 看起来"验证通过"其实是假的。⇒ **必须用 curl(`-H Host:`)**,且判据不能只看状态码。
*
* 运行(在 47 上):MANAGER=http://127.0.0.1:13080 BASE_DOMAIN=test.alotbuy.com \
* AGENT=http://127.0.0.1:19000 AGENT_TOKEN=cross-machine-token \
* node scripts/verify-cluster-domain.mjs
*/
import { execFileSync } from 'node:child_process'
import { readFileSync } from 'node:fs'
const MANAGER = process.env.MANAGER ?? 'http://127.0.0.1:13080'
const BASE_DOMAIN = process.env.BASE_DOMAIN ?? 'test.alotbuy.com'
const AGENT = process.env.AGENT ?? 'http://127.0.0.1:19000'
const TOKEN = process.env.AGENT_TOKEN ?? 'cross-machine-token'
const ADMIN_PW = process.env.ADMIN_PW ?? 'crossmgr123'
const USER_PW = 'domainuser123'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(MANAGER + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
signal: AbortSignal.timeout(30_000),
})
const text = await res.text()
let parsed = null
try {
parsed = text === '' ? null : JSON.parse(text)
} catch {
parsed = { raw: text.slice(0, 200) }
}
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
}
/**
* 用 **curl** 带 `Host` 头取页面(`-L` 跟随重定向 ⇒ 等价真实浏览器)。
* 返回 `{ status, body }`,body 从临时文件读(避免编码/二进制问题)。
*/
function getByHost(sub, path, cookie) {
const out = '/tmp/dompage.html'
const args = [
'-s',
'-L',
'--max-time',
'25',
'-o',
out,
'-w',
'%{http_code}',
'-H',
`Host: ${sub}.${BASE_DOMAIN}`,
...(cookie ? ['-b', cookie] : []),
`${MANAGER}${path}`,
]
let status = '0'
try {
status = execFileSync('curl', args, { encoding: 'utf8' }).trim()
} catch {
status = '0'
}
let body = ''
try {
body = readFileSync(out, 'utf8')
} catch {
body = ''
}
return { status: Number(status), body }
}
/**
* 判据:**dsh 实例页**带 `<base href="/">`(子路径与子域两种形态都带);平台门户页不带。
* ⚠️ 只靠"含 dsh 字样"会把门户页误判成实例页(实测踩过这个假阳性)。
*/
const isDshApp = (html) => typeof html === 'string' && html.includes('<base href=')
const describe = (html) => {
const hit = []
if (isDshApp(html)) hit.push('base-href')
if (html.includes('/api/auth/login')) hit.push('platform-login')
const t = /<title>([^<]*)<\/title>/.exec(html)
return `${hit.join(',') || '(无特征)'} | title=${t === null ? '?' : t[1].trim()} | 首100字: ${html.replace(/\s+/g, ' ').slice(0, 100)}`
}
async function waitRunning(cookie, tries = 60) {
for (let i = 0; i < tries; i += 1) {
const st = await json('/api/dsh/status', { cookie })
if (st.body?.running === true) return true
await sleep(1000)
}
return false
}
try {
console.log('=== 域名形态验证:baseDomain=%s(Manager=%s)===', BASE_DOMAIN, MANAGER)
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
assert(adm.status === 200, `管理员登录失败 ${adm.status}`)
const adminCookie = adm.setCookie.split(';')[0]
const uname = `domuser${Date.now() % 100000}`
let r = await json('/api/auth/register', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(r.status === 201, `注册应 201(实际 ${r.status})`)
const users = await json('/api/admin/users', { cookie: adminCookie })
const target = users.body.users.find((u) => u.username === uname)
assert(target !== undefined, '能看到待审用户')
await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
const login = await json('/api/auth/login', { method: 'POST', body: { username: uname, password: USER_PW } })
assert(login.status === 200, `用户登录失败 ${login.status}`)
const cookie = login.setCookie.split(';')[0]
console.log('① 用户 -> %s(uid=%s)', uname, target.id)
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 失败 ${r.status}`)
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 失败 ${r.status} ${JSON.stringify(r.body)}`)
assert(await waitRunning(cookie), '实例应 running')
console.log('② 拉起 -> running=true')
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter.status === 200, `enter 失败 ${enter.status}`)
const url = enter.body.url
console.log('③ 直达 URL -> %s', url)
assert(url.startsWith('https://'), `baseDomain 生效时应为 https://<子域>/(实际 ${url})`)
assert(url.includes(`${uname}.${BASE_DOMAIN}`), `URL 应含用户名子域(实际 ${url})`)
// ④ 子域形态访问:**必须带 token**(真 dsh 没 token 只给自己的登录页)
const token = new URL(url).searchParams.get('token') ?? ''
assert(token !== '', `enter URL 应带 token(实际 ${url})`)
let page = { status: 0, body: '' }
for (let i = 0; i < 40; i += 1) {
page = getByHost(uname, `/?token=${encodeURIComponent(token)}`, cookie)
if (page.status === 200 && isDshApp(page.body)) break
await sleep(1000)
}
assert(page.status === 200, `子域访问应 200(实际 ${page.status})`)
assert(isDshApp(page.body), `子域访问必须是**真的 dsh 实例页**(实际 ${describe(page.body)})`)
console.log('④ 子域访问 -> 200 且是**真 dsh 实例页**(Host: %s.%s → 106 上的实例)', uname, BASE_DOMAIN)
// ⑤ 越权对照:拿 A 的 cookie 访问**另一个真实用户**(root)的子域 ⇒ 必须 401/403
const other = getByHost('root', `/?token=${encodeURIComponent(token)}`, cookie)
assert(!isDshApp(other.body), `越权响应绝不能是实例页(实际 ${describe(other.body)})`)
assert([401, 403].includes(other.status), `用 A 的 cookie 访问 root 子域应 401/403(实际 ${other.status})`)
console.log('⑤ 越权对照 -> 用 A 的 cookie 访问 root 子域 = %d(正确拒绝)', other.status)
// ⑥ 独立取证:实例确实在 106
const onAgent = await fetch(`${AGENT}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } }).then((x) => x.json())
assert(onAgent.instances.length >= 1, 'worker(106) 上应有实例')
console.log('⑥ 取证 -> 实例确实在 106(worker /instances=%d)', onAgent.instances.length)
await json('/api/dsh/stop', { method: 'POST', cookie })
console.log('\nOK: **域名形态访问**在 cluster 下可用(子域 → Manager(47) → 实例(106)),且越权被拒')
} finally {
/* 现场保留 */
}
+155
View File
@@ -0,0 +1,155 @@
/**
* T08 S5 · 跨机文件面验证。
*
* 关键设计:**Manager 的 `dataRoot` 故意与 worker 的 `dataRoot` 不同** ——
* 只有这样"文件面真的走了远端"才被证明;若两个 root 相同,本地实现也能碰巧通过。
*
* 验的是:
* ① 门户的路由(`/api/desktop/tree`、`/api/fs/*`)在 cluster 模式下照常工作(**路由零改动**);
* ② 文件**落在 worker 的 dataRoot 下**、且**不在** Manager 的 dataRoot 下;
* ③ 路径安全与本地**同源**(`bad_path` 走同一条 `resolveWithinRoot`);
* ④ `resolvePath` 返回的是**实例眼里的路径**(按 worker 的 dataRoot 算)。
*
* 运行:node scripts/verify-cluster-fs.mjs
*/
import { existsSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
const TOKEN = 'verify-cluster-fs-token'
const workerRoot = mkdtempSync(join(tmpdir(), 'dsh-cfs-worker-'))
const managerRoot = mkdtempSync(join(tmpdir(), 'dsh-cfs-manager-'))
let agentApp
let agentHandle
let app
try {
// ── worker agent(dataRoot = workerRoot)────────────────────────────────
const agent = buildWorkerAgent(
resolveConfig({ port: 0, dbPath: ':memory:', dataRoot: workerRoot, dshCommand: [process.execPath, fakeDsh], clusterHostId: 'w-1' }),
{ hostId: 'w-1', token: TOKEN, port: 0, host: '127.0.0.1', instanceHost: '127.0.0.1', logLevel: 'warn' },
)
agentApp = agent.app
agentHandle = agent
await agentApp.listen({ host: '127.0.0.1', port: 0 })
const agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
console.log('worker -> dataRoot %s', workerRoot)
// ── Manager(dataRoot = managerRoot ≠ workerRoot;显式告知 worker 的 root)──
app = await buildServer(
resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot: managerRoot,
deployMode: 'cluster',
clusterAgentUrl: agentUrl,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
clusterHostId: 'm-1',
clusterWorkerDataRoot: workerRoot,
}),
)
await app.listen({ port: 0 })
const base = `http://127.0.0.1:${app.server.address().port}`
console.log('manager -> dataRoot %s(与 worker 不同 ⇒ 能证明走远端)', managerRoot)
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
}
await app.db.createUser({
id: 'u1',
username: 'bob',
passHash: await hashPassword('bobpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
// 用户根必须建在 **worker 上**(这一步本身就走远端)
await app.userFs.initUserRoot('u1')
// ④ resolvePath = 实例眼里的路径(按 worker 的 dataRoot)
const resolved = app.userFs.resolvePath('u1', 'proj')
assert(resolved === join(workerRoot, 'users', 'u1', 'ws', 'proj'), `resolvePath 应按 worker 的 root 计算(实际 ${resolved})`)
console.log('④ resolvePath -> %s', resolved)
// ── ① 门户路由(零改动)──────────────────────────────────────────────
let r = await json('/api/auth/login', { method: 'POST', body: { username: 'bob', password: 'bobpass123' } })
assert(r.status === 200, 'login succeeds')
const cookie = r.setCookie.split(';')[0]
r = await json('/api/desktop/tree', { cookie })
assert(r.status === 200 && r.body.entries.length === 0, '空工作区列出 0 项')
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 经远端成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
r = await json('/api/fs/upload', {
method: 'POST',
cookie,
body: { path: 'proj', name: 'hello.txt', data: Buffer.from('hi there').toString('base64') },
})
assert(r.status === 200, `upload 经远端成功(实际 ${r.status})`)
r = await json('/api/desktop/tree', { cookie })
assert(r.status === 200 && r.body.entries.length === 1, '工作区里出现了 proj')
console.log('① 门户路由 -> tree/mkdir/upload 全部经远端通过')
// ── ② 文件真的落在 worker 上 ──────────────────────────────────────────
const onWorker = join(workerRoot, 'users', 'u1', 'ws', 'proj', 'hello.txt')
const onManager = join(managerRoot, 'users', 'u1', 'ws', 'proj', 'hello.txt')
assert(existsSync(onWorker), `文件应落在 worker:${onWorker}`)
assert(!existsSync(onManager), `文件不该出现在 Manager 本地:${onManager}`)
console.log('② 落点 -> worker 有、manager 无(确认走远端)')
// ── ③ 路径安全与本地同源 ──────────────────────────────────────────────
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: '../evil' } })
assert(r.status === 400 && r.body.error === 'bad_path', `越界路径应 400 bad_path(实际 ${r.status} ${JSON.stringify(r.body)})`)
for (const bad of ['..', '../../etc']) {
let threw = false
try {
app.userFs.resolvePath('u1', bad)
} catch (err) {
threw = err.code === 'bad_path'
}
assert(threw, `resolvePath(${bad}) 应抛 bad_path`)
}
console.log('③ 路径安全 -> bad_path 与本地同源(走同一个 resolveWithinRoot)')
// 下载回读(readFile 经远端)—— 注意该路由回的是**原始字节**,不是 JSON
const dl = await fetch(`${base}/api/fs/download?path=proj/hello.txt`, { headers: { cookie } })
assert(dl.status === 200, `download 经远端成功(实际 ${dl.status})`)
const downloaded = await dl.text()
assert(downloaded === 'hi there', `下载内容应为上传的原文(实际 ${JSON.stringify(downloaded)})`)
console.log(' 下载回读 -> readFile 经远端成功(内容逐字一致)')
console.log('\nOK: 跨机文件面(RemoteUserFs → agent /fs/*)通过')
console.log(' ✓ 门户路由零改动 ✓ 落在 worker ✓ 路径安全同源 ✓ resolvePath 按 worker 计算')
} finally {
await app?.close()
await agentHandle?.stop()
await new Promise((r) => setTimeout(r, 300))
for (const dir of [workerRoot, managerRoot]) {
try {
rmSync(dir, { recursive: true, force: true })
} catch {
/* best-effort */
}
}
}
+208
View File
@@ -0,0 +1,208 @@
/**
* T08 S4 · 归属租约端到端验证(1a 形态:多 Manager + 一个 worker agent + 共享 PG)。
*
* 验的是 S4 的四条承重行为:
* ① **归属真的落库**:launch 后 `dsh_instances` 有 `host_id` / `epoch` / `lease_until`;
* ② **单写者**:另一个 Manager(另一个 worker 身份)在租约存活期内**拉不起来**同一用户
* ⇒ 抛 `LeaseBusyError`(退让,不是接管);
* ③ **stop 释放归属** ⇒ 别人立刻能接(不用等 TTL);
* ④ **失权即 self-fence**:租约被抢走后,原持有者下一次心跳会把"更高 epoch"下发给 worker,
* 由 worker **停掉自己那个实例**(防双写的最后一道防线)。
* ⑤ 顺带验**注册 + 心跳**:`dsh_hosts` 里有记录且 `last_heartbeat` 持续更新。
*
* 需要 PG(两个 Manager 必须共享 DB,否则谈不上"归属"):
* CLUSTER_TEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node scripts/verify-cluster-lease.mjs
*/
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import pg from 'pg'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const PG_URL = process.env.CLUSTER_TEST_PG_URL
if (PG_URL === undefined || PG_URL === '') {
console.error('需要 CLUSTER_TEST_PG_URL(两个 Manager 必须共享同一个库)')
process.exit(2)
}
const TOKEN = 'verify-cluster-lease-token'
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
const dataRoot = mkdtempSync(join(tmpdir(), 'dsh-lease-'))
// 短 TTL:TTL=1200ms > 2×renew=500ms(满足不变量),便于在秒级制造"过期/被抢"
process.env.DSHS_CLUSTER_LEASE_TTL_MS = '1200'
process.env.DSHS_CLUSTER_LEASE_RENEW_MS = '500'
let agentApp
let agentHandle
const managers = []
/** 清空测试库里的三张表(该库专供本测试)。 */
async function resetPg() {
const client = new pg.Client({ connectionString: PG_URL })
await client.connect()
await client.query('DELETE FROM dsh_instances')
await client.query('DELETE FROM dsh_hosts')
await client.query('DELETE FROM users')
await client.end()
}
/** 起一个 Manager(cluster 模式)。hostId 即"它绑定的 worker 身份"。 */
async function startManager(hostId) {
const app = await buildServer(
resolveConfig({
port: 0,
dbUrl: PG_URL,
dataRoot,
deployMode: 'cluster',
clusterAgentUrl: agentUrl,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
clusterHostId: hostId,
}),
)
await app.listen({ port: 0 })
managers.push(app)
return app
}
let agentUrl = ''
try {
await resetPg()
// ── 0) worker agent ───────────────────────────────────────────────────
const agentConfig = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot,
dshCommand: [process.execPath, fakeDsh],
clusterHostId: 'w-1',
})
const agent = buildWorkerAgent(agentConfig, {
hostId: 'w-1',
token: TOKEN,
port: 0,
host: '127.0.0.1',
instanceHost: '127.0.0.1',
logLevel: 'warn',
})
agentApp = agent.app
agentHandle = agent
await agentApp.listen({ host: '127.0.0.1', port: 0 })
agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
const agentJson = async (path, { method = 'GET', body } = {}) => {
const res = await fetch(agentUrl + path, {
method,
headers: { [AGENT_TOKEN_HEADER]: TOKEN, ...(body ? { 'content-type': 'application/json' } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null }
}
// ── 1) 两个 Manager(不同 worker 身份)────────────────────────────────
const m1 = await startManager('m-1')
const m2 = await startManager('m-2')
console.log('manager -> m-1 %s / m-2 %s(共享 PG + 同一 agent)', m1.supervisor.hostId, m2.supervisor.hostId)
await m1.db.createUser({
id: 'u1',
username: 'carol',
passHash: await hashPassword('carolpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
mkdirSync(join(dataRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
const folder = join(dataRoot, 'users', 'u1', 'ws', 'proj')
// ⑤ 注册 + 心跳:dsh_hosts 里应有记录(启动即注册 + 立即一次心跳)
const hosts = await m1.db.listDshHosts()
assert(hosts.length === 2, `dsh_hosts 应有 2 条(实际 ${hosts.length})`)
assert(hosts.every((h) => h.status === 'up' && h.lastHeartbeat !== null), 'worker 状态 up 且有心跳时间')
console.log('注册/心跳 -> dsh_hosts =', hosts.map((h) => `${h.id}:${h.status}`).join(', '))
// ── 2) m-1 拉起:归属必须落库 ─────────────────────────────────────────
const inst1 = await m1.supervisor.launch('u1', folder)
assert(inst1.userId === 'u1', 'launch 返回实例')
let row = await m1.db.findUserInstance('u1', 'main')
assert(row.hostId === 'm-1', `归属应落库为 m-1(实际 ${row.hostId})`)
assert(row.epoch === 1, `首次抢占 epoch 应为 1(实际 ${row.epoch})`)
assert(row.leaseUntil > Date.now(), 'lease_until 应在未来')
console.log('① 归属落库 -> host_id=%s epoch=%d lease_until=+%dms', row.hostId, row.epoch, row.leaseUntil - Date.now())
// worker 侧真的收到了 epoch=1(`/fence` 同值 ⇒ 不该被 fence)
const f0 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 1 } })
assert(f0.body.fenced === false, 'worker 已记录 epoch=1(同值不 fence)')
console.log(' worker 已记录 epoch=1')
// ── 3) 单写者:m-2 在租约存活期内拉不起来 ────────────────────────────
let busy
try {
await m2.supervisor.launch('u1', folder)
} catch (err) {
busy = err
}
assert(busy !== undefined, 'm-2 必须拉起失败')
assert(busy.name === 'LeaseBusyError', `应是 LeaseBusyError(实际 ${busy.name})`)
assert(busy.holder === 'm-1', `错误里应带持有者 m-1(实际 ${busy.holder})`)
const stillMine = await m1.db.findUserInstance('u1', 'main')
assert(stillMine.hostId === 'm-1' && stillMine.epoch === 1, '失败方不得改动归属')
console.log('② 单写者 -> m-2 抛 LeaseBusyError(holder=%s),归属未被改动', busy.holder)
// ── 4) stop 释放 ⇒ 别人立刻能接(不用等 TTL)──────────────────────────
await m1.supervisor.stop('u1')
row = await m1.db.findUserInstance('u1', 'main')
assert(row.hostId === null, 'stop 后归属应清空')
// 多机语义(S6 起):不显式指定目标机时由 `selectHost` 按容量挑"最优的那台",
// 不一定是 m-2 ⇒ 这一步要验的是"释放后可被接管",所以**显式指定 m-2**(确定性)。
const inst2 = await m2.supervisor.launch('u1', folder, undefined, { hostId: 'm-2' })
assert(inst2.userId === 'u1', 'm-2 拉起成功')
row = await m2.db.findUserInstance('u1', 'main')
assert(row.hostId === 'm-2', `归属应转给 m-2(实际 ${row.hostId})`)
assert(row.epoch === 2, `epoch 必须递增到 2(实际 ${row.epoch})`)
console.log('③ 释放即接手 -> host_id=m-2 epoch=%d(epoch 单调递增)', row.epoch)
// ── 5) 失权即 self-fence ─────────────────────────────────────────────
// 让 m-2 也"死掉"(停心跳)→ 等待 TTL 过期 → m-1 抢占(epoch=3)
m2.supervisor.stopHeartbeat()
await sleep(1400)
const stolen = await m1.db.claimInstance('u1', 'm-1', 60_000)
assert(stolen.ok === true && stolen.epoch === 3, `m-1 过期后应能抢到 epoch=3(实际 ${JSON.stringify(stolen)})`)
console.log(' m-1 在租约过期后抢回(epoch=3)')
// m-2 的下一跳心跳发现自己失权 ⇒ 给 worker 下发更高 epoch ⇒ worker 停掉自己那个实例
const before = await agentJson('/instances')
await m2.supervisor.tick()
await sleep(200)
const after = await agentJson('/instances')
assert(after.body.instances.length === 0, `失权方心跳后实例应被停(前 ${before.body.instances.length} → 后 ${after.body.instances.length})`)
console.log('④ 失权即 fence -> worker 实例数 %d → %d(self-fencing 生效)', before.body.instances.length, after.body.instances.length)
// 归属仍在 m-1 名下(fence 不会误清他人的归属)
row = await m1.db.findUserInstance('u1', 'main')
assert(row.hostId === 'm-1', 'fence 不该清掉持有者的归属')
console.log(' 归属仍在 m-1 名下(未被误清)')
console.log('\nOK: 归属租约(1a 形态)端到端通过')
console.log(' ✓ 归属落库 ✓ 单写者(LeaseBusyError) ✓ 释放即接手 ✓ 失权即 self-fence ✓ 注册+心跳')
} finally {
for (const app of managers) await app?.close()
await agentHandle?.stop()
await sleep(500)
try {
rmSync(dataRoot, { recursive: true, force: true })
} catch {
/* best-effort */
}
}
+365
View File
@@ -0,0 +1,365 @@
/**
* T08 · **真实部署**端到端功能确认(不是单进程内测试)。
*
* 与 `verify-cluster-*.mjs` 的区别(那些是**组件级**验证,两个 Fastify 跑在同一进程里):
* 这里**真的起进程** —— 2 个 `dshs worker` agent + 1 个 Manager 都是独立进程,
* 经**真 HTTP**(127.0.0.1 端口)与**真 PG** 通信,实例是**真 `dsh` 子进程**(account 隔离)。
* 它回答的是最后一个问题:**这套东西按部署形态装起来,到底能不能用。**
*
* 检查链路(一条真实的用户路径):
* ① bootstrap-admin → ③ register → approve(平台现有流程)
* ④ 登录 → ⑤ 建文件夹(经 RemoteUserFs 落到 worker)→ ⑥ launch(经 agent 起真 dsh)
* ⑦ 轮询 status → ⑧ **取实例页面(经代理)**← 功能确认的关键一步
* ⑨ stop → ⑩ 起第二台 worker → 注册 → **迁移** → 再取一次页面
* ⑪ `dshs doctor` / `dshs cluster status`(观测面)
*
* 需要:106 上 PG 已在 127.0.0.1:15432;以 root 运行(account 隔离要 setpriv/systemd-run)。
* 运行:CLUSTER_LIVE_PG_URL=postgres://dshs:[email protected]:15432/dshs_live node scripts/verify-cluster-live.mjs
*/
import { spawn, spawnSync } from 'node:child_process'
import { createWriteStream, mkdirSync, readFileSync, rmSync } from 'node:fs'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const PG_URL = process.env.CLUSTER_LIVE_PG_URL
if (PG_URL === undefined || PG_URL === '') {
console.error('需要 CLUSTER_LIVE_PG_URL')
process.exit(2)
}
const here = dirname(fileURLToPath(import.meta.url))
const repoRoot = join(here, '..')
const CLI = join(repoRoot, 'lib', 'cli.js')
const TOKEN = 'live-cluster-agent-token'
const DATA_ROOT = process.env.CLUSTER_LIVE_DATA_ROOT ?? '/opt/dshs-cluster/live-data'
const ISO = process.env.CLUSTER_LIVE_ISOLATION ?? 'account'
const M_PORT = Number(process.env.CLUSTER_LIVE_MANAGER_PORT ?? 13080)
const A1_PORT = Number(process.env.CLUSTER_LIVE_AGENT1_PORT ?? 19000)
const A2_PORT = Number(process.env.CLUSTER_LIVE_AGENT2_PORT ?? 19001)
const ADMIN_PW = 'liveadmin123'
const USER_PW = 'liveuser123'
const procs = []
/** 起一个子进程并记下来(收尾统一 SIGTERM ⇒ agent 会先 teardown 实例再退出)。 */
function run(label, args, env) {
const child = spawn(process.execPath, args, {
cwd: repoRoot,
env: { ...process.env, ...env },
stdio: ['ignore', 'pipe', 'pipe'],
})
// ⚠️ **必须留日志**:不留就只能看到"实例没了"而看不到为什么(2026-09-15 实测踩到)
const logPath = `/tmp/live-${label}.log`
const stream = createWriteStream(logPath, { flags: 'w' })
child.stdout.pipe(stream)
child.stderr.pipe(stream)
child.on('exit', (code) => {
if (code !== null && code !== 0 && !stopping) console.error(`[${label}] 提前退出 code=${code}(日志 ${logPath})`)
})
procs.push({ label, child, logPath })
return child
}
/** 打印某个子进程日志的尾部(诊断用)。 */
function tailLog(label, lines = 12) {
const found = procs.find((p) => p.label === label)
if (found === undefined) return
try {
const text = readFileSync(found.logPath, 'utf8').trimEnd().split('\n')
console.error(` ── ${label} 日志尾部 ──`)
for (const line of text.slice(-lines)) console.error(' ', line.slice(0, 200))
} catch {
/* 没日志就算了 */
}
}
let stopping = false
function shutdownAll() {
stopping = true
for (const { child } of procs) {
try {
child.kill('SIGTERM')
} catch {
/* 已退出 */
}
}
}
/** 轮询等一个 URL 可用。 */
async function waitHttp(url, timeoutMs, what) {
const deadline = Date.now() + timeoutMs
let last = ''
while (Date.now() < deadline) {
try {
const res = await fetch(url, { signal: AbortSignal.timeout(2000) })
if (res.status < 500) return res.status
last = `HTTP ${res.status}`
} catch (err) {
last = err instanceof Error ? err.message : String(err)
}
await sleep(300)
}
throw new Error(`等待 ${what} 超时(${url}):${last}`)
}
const base = `http://127.0.0.1:${M_PORT}`
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
signal: AbortSignal.timeout(30_000),
})
const text = await res.text()
let parsed = null
try {
parsed = text === '' ? null : JSON.parse(text)
} catch {
parsed = { raw: text.slice(0, 200) }
}
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
}
try {
console.log('=== 真实部署检查:dataRoot=%s 隔离=%s ===', DATA_ROOT, ISO)
rmSync(DATA_ROOT, { recursive: true, force: true })
mkdirSync(DATA_ROOT, { recursive: true })
// 干净的 PG 库(本检查专用)
const psql = (sql) =>
spawnSync('su', ['-', 'postgres', '-c', `/usr/bin/psql -p 15432 -q -c "${sql}"`], { encoding: 'utf8' })
psql('DROP DATABASE IF EXISTS dshs_live')
psql('CREATE DATABASE dshs_live OWNER dshs')
console.log('PG -> dshs_live 已重建')
// ── ① worker agent(独立进程)─────────────────────────────────────────
const agentEnv = { DSHS_DATA_ROOT: DATA_ROOT, DSHS_ISOLATION_MODE: ISO }
run('agent-w-1', [CLI, 'worker', '--token', TOKEN, '--port', String(A1_PORT), '--host', '127.0.0.1', '--host-id', 'w-1', '--instance-host', '127.0.0.1', '--log-level', 'warn'], agentEnv)
await waitHttp(`http://127.0.0.1:${A1_PORT}/healthz`, 15_000, 'agent w-1')
console.log('worker w-1 -> http://127.0.0.1:%d 就绪', A1_PORT)
// ── ② Manager(独立进程,cluster 模式)────────────────────────────────
const managerEnv = {
DSHS_DEPLOY_MODE: 'cluster',
DSHS_DB_URL: PG_URL,
DSHS_DATA_ROOT: DATA_ROOT,
DSHS_CLUSTER_HOST_ID: 'm-1',
DSHS_CLUSTER_AGENT_URL: `http://127.0.0.1:${A1_PORT}`,
DSHS_CLUSTER_AGENT_TOKEN: TOKEN,
DSHS_CLUSTER_INSTANCE_HOST: '127.0.0.1',
DSHS_CLUSTER_WORKER_DATA_ROOT: DATA_ROOT,
DSHS_CLUSTER_CAPACITY_MB: '-1', // Manager 自己**不承载实例**
DSHS_CLUSTER_REGISTER_SELF: '0', // 专用 Manager ⇒ **不自注册**(一个 agent 只应有一条 host 记录)
DSHS_CLUSTER_LEASE_TTL_MS: '30000',
}
run('manager', [CLI, '--port', String(M_PORT), '--host', '127.0.0.1', '--log-level', 'warn'], managerEnv)
await waitHttp(`${base}/login.html`, 20_000, 'Manager')
console.log('manager -> %s 就绪(deployMode=cluster)', base)
// ── ③ bootstrap-admin(首次建管理员)──────────────────────────────────
const boot = spawnSync(process.execPath, [CLI, 'bootstrap-admin', '--username', 'root', '--password', ADMIN_PW], {
cwd: repoRoot,
// 用 local 模式初始化管理员 root:它就是这台机上的目录,与 worker 用同一个 DATA_ROOT
env: { ...process.env, DSHS_DATA_ROOT: DATA_ROOT, DSHS_DB_URL: PG_URL },
encoding: 'utf8',
})
assert(boot.status === 0, `bootstrap-admin 失败:${boot.stderr?.slice(0, 300)}`)
console.log('管理员 -> root 已创建(bootstrap-admin)')
// ── ④ 真实用户流程:注册 → 审批 → 登录 ────────────────────────────────
let adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
assert(adm.status === 200, `管理员登录失败:${adm.status}`)
const adminCookie = adm.setCookie.split(';')[0]
let r = await json('/api/auth/register', { method: 'POST', body: { username: 'liveuser', password: USER_PW } })
assert(r.status === 201, `注册应 201(实际 ${r.status} ${JSON.stringify(r.body)})`)
const users = await json('/api/admin/users', { cookie: adminCookie })
const target = users.body.users.find((u) => u.username === 'liveuser')
assert(target !== undefined, '管理员能列出待审用户')
r = await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
assert(r.status === 200, `审批应 200(实际 ${r.status})`)
const login = await json('/api/auth/login', { method: 'POST', body: { username: 'liveuser', password: USER_PW } })
assert(login.status === 200, `用户登录失败:${login.status}`)
const cookie = login.setCookie.split(';')[0]
console.log('① 用户流程 -> 注册 → 审批 → 登录 全部通过(uid=%s)', target.id)
// 显式注册 w-1(= join 脚本那一步:agent 已在跑,调管理面登记)
r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-1', endpoint: `http://127.0.0.1:${A1_PORT}`, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-1 应 200(实际 ${r.status})`)
// ── ⑤ 建文件夹(经 RemoteUserFs 落到 worker)─────────────────────────
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
assert(r.status === 200, `mkdir 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(
(await json('/api/desktop/tree', { cookie })).body.entries.some((e) => e.name === 'proj'),
'工作区里出现 proj',
)
console.log('② 文件面 -> mkdir 落库到 worker(经 agent /fs/mkdir)')
// ── ⑥ 拉起实例(经 agent 起**真 dsh**)───────────────────────────────
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(typeof r.body.url === 'string' && r.body.url.startsWith('/u/'), `launch 返回子路径形态 URL(实际 ${r.body.url})`)
console.log('③ 拉起 -> %s(真 dsh 启动中,token 稍后才吐)', r.body.url)
// 直接问 worker(绕过 Manager):实例到底在不在 agent 手上
const onAgent = await fetch(`http://127.0.0.1:${A1_PORT}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } })
console.log(' worker 视角 -> /instances = %s', (await onAgent.text()).slice(0, 200))
// ── ⑦ 轮询到「running **且** launch token 到位」 ─────────────────────
// 真 dsh 与 fake-dsh 不同:进程起来 ≠ 已打印 token。**launch token 回传(P0-6)**
// 只有在 token 真的从 worker 传回 Manager 之后才算成立,所以这里要等到它。
const t0 = Date.now()
let running = false
let tokenSeen = ''
let lastStatus = null
for (let i = 0; i < 90; i += 1) {
const st = await json('/api/dsh/status', { cookie })
lastStatus = st.body
const main = st.body.main
if (main?.status === 'crashed') {
console.error(' 实例崩溃:exitCode=%s lastError=%s', main.exitCode, String(main.lastError).slice(0, 400))
break
}
// 注意:`/api/dsh/status` 的实例视图是**精简视图**(id/port/status/restarts),
// **不含 launchToken** ⇒ token 的存在性用下面的 `/api/dsh/enter` 判定(它回带 token 的 URL)。
if (st.body.running === true) {
running = true
tokenSeen = st.body.url ?? ''
break
}
await sleep(1000)
}
if (!running) {
tailLog('agent-w-1', 20)
tailLog('manager', 10)
}
assert(running, `实例应在 90s 内 running(实际 ${JSON.stringify(lastStatus)?.slice(0, 500)})`)
console.log('④ 状态 -> running=true(真 dsh,隔离=%s,耗时 %ds)', ISO, Math.round((Date.now() - t0) / 1000))
// ── ⑧ **登录直达**(P0-6 的真实端到端):enter 走"复用已运行实例"分支 ⇒ 带 token 的 URL
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
assert(enter.status === 200, `enter 应 200(实际 ${enter.status} ${JSON.stringify(enter.body)})`)
const launchUrl = enter.body.url
assert(typeof launchUrl === 'string' && launchUrl.includes('token='), `enter 应返回**带 token** 的直达 URL(实际 ${launchUrl})`)
console.log('⑤ 登录直达 -> %s', launchUrl)
let pageStatus = 0
let pageSnippet = ''
for (let i = 0; i < 40; i += 1) {
// 真实浏览器会**跟随重定向**(dsh 首页 303 → 应用页)⇒ 这里也跟随,否则会误判为失败。
// ⚠️ 必须 try/catch:实例刚 spawn 时正在初始化,代理可能中途断连(`other side closed`),
// 这是**启动窗口的正常现象**,重试即可 —— 不捕获会让检查在第一次尝试就失败。
try {
const res = await fetch(base + launchUrl, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(15_000) })
pageStatus = res.status
if (res.status === 200) {
pageSnippet = (await res.text()).slice(0, 400)
break
}
} catch {
pageStatus = 0
}
await sleep(1000)
}
assert(pageStatus === 200, `实例页面应最终 200(实际 ${pageStatus})`)
console.log('⑥ 实例页面 -> 200(经 Manager 代理到 worker 上 account 沙箱内的真 dsh;已跟随 303 重定向)')
// ── ⑨ 停止 ────────────────────────────────────────────────────────────
r = await json('/api/dsh/stop', { method: 'POST', cookie })
assert(r.status === 200, `stop 应 200(实际 ${r.status})`)
await sleep(500)
assert((await json('/api/dsh/status', { cookie })).body.running === false, 'stop 后 running=false')
console.log('⑦ 停止 -> ok')
// ── ⑩ 第二台 worker + 迁移 ────────────────────────────────────────────
run('agent-w-2', [CLI, 'worker', '--token', TOKEN, '--port', String(A2_PORT), '--host', '127.0.0.1', '--host-id', 'w-2', '--instance-host', '127.0.0.1', '--log-level', 'warn'], agentEnv)
await waitHttp(`http://127.0.0.1:${A2_PORT}/healthz`, 15_000, 'agent w-2')
r = await json('/api/admin/hosts', {
method: 'POST',
cookie: adminCookie,
body: { id: 'w-2', endpoint: `http://127.0.0.1:${A2_PORT}`, token: TOKEN, capacityMb: 4096 },
})
assert(r.status === 200, `注册 w-2 应 200(实际 ${r.status})`)
const hosts = await json('/api/admin/hosts', { cookie: adminCookie })
assert(hosts.body.hosts.some((h) => h.id === 'w-2'), 'w-2 出现在 worker 目录')
assert(!('agentToken' in (hosts.body.hosts[0] ?? {})), '**绝不下发 agentToken**')
console.log('⑧ 第二台 -> w-2 已注册(且列表不含 agentToken)')
// 重新拉起(此刻只有 w-1 是候选 ⇒ 确定性落 w-1),再注册 w-2、再迁移
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
assert(r.status === 200, `再次 launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
for (let i = 0; i < 40 && !(await json('/api/dsh/status', { cookie })).body.running; i += 1) await sleep(1000)
const owner = (await json('/api/dsh/status', { cookie })).body
assert(owner.running === true, '重新拉起后 running=true')
r = await json(`/api/admin/users/${target.id}/dsh/migrate`, {
method: 'POST',
cookie: adminCookie,
body: { targetHost: 'w-2' },
})
assert(r.status === 200, `迁移应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(r.body.to === 'w-2', `迁移目标应为 w-2(实际 ${r.body.to})`)
for (let i = 0; i < 30 && !(await json('/api/dsh/status', { cookie })).body.running; i += 1) await sleep(1000)
console.log('⑨ 迁移 -> %s → %s(epoch=%d)', r.body.from, r.body.to, r.body.epoch)
// ⚠️ 迁移后实例是**新进程 ⇒ 新 launch token**:旧 URL 里的 token 已失效(404 是**预期**行为)。
// 真实用户会重新走 `/api/dsh/enter`(门户的"进入工作区"就是这个接口)拿**新** URL ⇒ 这里照做。
let newUrl = ''
pageStatus = 0
for (let i = 0; i < 40; i += 1) {
try {
const en = await json('/api/dsh/enter', { method: 'POST', cookie })
if (en.status === 200 && typeof en.body.url === 'string') {
newUrl = en.body.url
const res = await fetch(base + newUrl, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(15_000) })
pageStatus = res.status
if (res.status === 200) {
pageSnippet = (await res.text()).slice(0, 200)
break
}
}
} catch {
pageStatus = 0
}
await sleep(1000)
}
if (newUrl === '' || newUrl === launchUrl) {
// 诊断:两台 agent 各自认为有什么 + DB 归属如何
for (const [label, port] of [['w-1', A1_PORT], ['w-2', A2_PORT]]) {
const res = await fetch(`http://127.0.0.1:${port}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } })
const body = await res.text()
console.error(` ${label} /instances = ${body.slice(0, 220)}`)
const st = await fetch(`http://127.0.0.1:${port}/status/${target.id}`, { headers: { 'x-dsh-agent-token': TOKEN } })
console.error(` ${label} /status = ${(await st.text()).slice(0, 220)}`)
}
tailLog('agent-w-2', 16)
tailLog('manager', 8)
}
assert(newUrl !== '' && newUrl !== launchUrl, `迁移后 enter 应给**新** URL(旧 ${launchUrl} / 新 ${newUrl})`)
assert(pageStatus === 200, `迁移后经新 URL 的实例页面应 200(实际 ${pageStatus})`)
console.log('⑩ 迁移后 -> enter 返回新 token URL,页面 200(经 w-2)')
// ── ⑪ 观测面 ──────────────────────────────────────────────────────────
const doctor = spawnSync(process.execPath, [CLI, 'doctor'], { cwd: repoRoot, env: { ...process.env, ...managerEnv }, encoding: 'utf8' })
const status = spawnSync(process.execPath, [CLI, 'cluster', 'status'], { cwd: repoRoot, env: { ...process.env, ...managerEnv }, encoding: 'utf8' })
console.log('⑪ dshs doctor -> rc=%d(0 = 无硬失败)', doctor.status ?? -1)
console.log(String(status.stdout).split('\n').slice(0, 8).map((l) => ' ' + l).join('\n'))
// 收尾:停实例(避免留下 dsh 子进程)
await json('/api/dsh/stop', { method: 'POST', cookie })
console.log('\nOK: 真实部署(2 个 worker agent 进程 + 1 个 Manager 进程 + 真 PG)端到端功能确认通过')
console.log(' ✓ 用户流程 ✓ 文件面跨机 ✓ 真 dsh 拉起并可从公网侧取页面 ✓ 停止 ✓ 注册+迁移+迁移后复验 ✓ 观测面')
console.log(' 页面片段:%s', pageSnippet.replace(/\s+/g, ' ').slice(0, 80))
} finally {
shutdownAll()
await sleep(1500)
}
+246
View File
@@ -0,0 +1,246 @@
/**
* T08 S6 · 多 worker + 容量准入 + **计划内迁移**验证。
*
* 这一条是整套设计的落点:**实例可迁移**。它同时验证:
* ① **容量准入**:`selectHost` 把"已用 + 预留 > 容量"的机排除掉 ⇒ 实例落到还有余量的那台;
* ② **归属与实例一致**:`dsh_instances.host_id` 指向实例真正所在的那台;
* ③ **迁移三步**(drain → 目标机拉起 → 归属原子更新):`host_id` 换台、`epoch` 单调 +1;
* ④ **迁移后代理照常**:`endpointFor` 按新归属路由,页面仍 200;
* ⑤ **数据不搬家也能用**:两台 worker **共享同一 dataRoot**(模拟共享存储 / 同路径基线)。
*
* 需要 PG(归属在 DB 里,两个 Manager/worker 共享):
* CLUSTER_TEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node scripts/verify-cluster-migrate.mjs
*/
import { existsSync, mkdirSync, mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import pg from 'pg'
import { buildServer } from '../lib/web/server.js'
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
import { resolveConfig } from '../lib/config.js'
import { hashPassword } from '../lib/web/auth.js'
function assert(condition, message) {
if (!condition) throw new Error('ASSERT: ' + message)
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
const PG_URL = process.env.CLUSTER_TEST_PG_URL
if (PG_URL === undefined || PG_URL === '') {
console.error('需要 CLUSTER_TEST_PG_URL')
process.exit(2)
}
const TOKEN = 'verify-cluster-migrate-token'
const here = dirname(fileURLToPath(import.meta.url))
const fakeDsh = join(here, 'fake-dsh.mjs')
/** 两台 worker **共享同一 dataRoot** = 模拟共享存储 / "所有 worker 同路径"的基线约定。 */
const sharedRoot = mkdtempSync(join(tmpdir(), 'dsh-migrate-'))
// Manager 自己不承载实例(capacity=-1);两台 worker 声明 4096MB
process.env.DSHS_CLUSTER_CAPACITY_MB = '-1'
let app
const agents = []
async function resetPg() {
const client = new pg.Client({ connectionString: PG_URL })
await client.connect()
await client.query('DELETE FROM dsh_instances')
await client.query('DELETE FROM dsh_hosts')
await client.query('DELETE FROM users')
await client.end()
}
async function startAgent(hostId) {
const config = resolveConfig({
port: 0,
dbPath: ':memory:',
dataRoot: sharedRoot,
dshCommand: [process.execPath, fakeDsh],
clusterHostId: hostId,
})
const agent = buildWorkerAgent(config, {
hostId,
token: TOKEN,
port: 0,
host: '127.0.0.1',
instanceHost: '127.0.0.1',
logLevel: 'warn',
})
await agent.app.listen({ host: '127.0.0.1', port: 0 })
agents.push(agent) // 整个 handle:收尾要用 stop() 收实例
const url = `http://127.0.0.1:${agent.app.server.address().port}`
const call = async (path, { method = 'GET', body } = {}) => {
const res = await fetch(url + path, {
method,
headers: { [AGENT_TOKEN_HEADER]: TOKEN, ...(body ? { 'content-type': 'application/json' } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null }
}
/** 该机上的实例数(对账口径)。 */
const instanceCount = async () => (await call('/instances')).body.instances.length
return { hostId, url, call, instanceCount }
}
try {
await resetPg()
// ── 0) 两台 worker ────────────────────────────────────────────────────
const a = await startAgent('w-a')
const b = await startAgent('w-b')
console.log('worker -> w-a %s / w-b %s(共享 dataRoot)', a.url, b.url)
// ── 1) Manager(默认 agent 指 w-a;自己 capacity=-1 不承载)─────────────
app = await buildServer(
resolveConfig({
port: 0,
dbUrl: PG_URL,
dataRoot: sharedRoot,
deployMode: 'cluster',
clusterAgentUrl: a.url,
clusterAgentToken: TOKEN,
clusterInstanceHost: '127.0.0.1',
clusterHostId: 'm-1',
clusterWorkerDataRoot: sharedRoot,
}),
)
await app.listen({ port: 0 })
const base = `http://127.0.0.1:${app.server.address().port}`
// 注册两台 worker(join 脚本走的就是这个 API)
await app.db.upsertDshHost({ id: 'w-a', endpoint: a.url, agentToken: TOKEN, capacityMb: 4096 })
await app.db.upsertDshHost({ id: 'w-b', endpoint: b.url, agentToken: TOKEN, capacityMb: 4096 })
// 把 w-b 的已用水位抬高到"再来一个实例就超" ⇒ 用来验证**准入拒绝**
await app.db.setDshHostStatus('w-b', 'up', 3800, Date.now())
// admin 账号(迁移 API 需要)
await app.db.createUser({
id: 'admin-1',
username: 'root',
passHash: await hashPassword('rootpass123'),
role: 'admin',
homeDir: '/tmp/admin-home',
})
await app.db.createUser({
id: 'u1',
username: 'carol',
passHash: await hashPassword('carolpass123'),
role: 'active',
homeDir: '/tmp/u1-home',
})
await app.userFs.initUserRoot('u1')
// 门户流程里 folder 是用户从「我的文件」里挑的**已存在**目录 ⇒ 这里先建出来
mkdirSync(join(sharedRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
const json = async (path, { method = 'GET', body, cookie } = {}) => {
const res = await fetch(base + path, {
method,
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
body: body ? JSON.stringify(body) : undefined,
})
const text = await res.text()
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
}
// ── 2) 容量准入:w-b 水位高 ⇒ 必须落到 w-a ────────────────────────────
const c = await json('/api/auth/login', { method: 'POST', body: { username: 'carol', password: 'carolpass123' } })
assert(c.status === 200, 'user login')
const userCookie = c.setCookie.split(';')[0]
let r = await json('/api/dsh/launch', { method: 'POST', cookie: userCookie, body: { folder: 'proj' } })
assert(r.status === 200, `launch 经远端成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
let row = await app.db.findUserInstance('u1', 'main')
assert(row.hostId === 'w-a', `容量准入应选 w-a(w-b 已 3800+512>4096);实际 ${row.hostId}`)
assert(row.epoch === 1, `首次抢占 epoch=1(实际 ${row.epoch})`)
assert((await a.instanceCount()) === 1, 'w-a 上有 1 个实例')
assert((await b.instanceCount()) === 0, 'w-b 上 0 个实例')
console.log('① 容量准入 -> 落到 w-a(w-b 因水位被排除),host_id=w-a epoch=1')
// 代理照常
let proxyText
for (let i = 0; i < 20; i += 1) {
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie: userCookie } })
if (res.status === 200) {
proxyText = await res.text()
break
}
await sleep(100)
}
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), '迁移前代理 200(经 w-a)')
console.log(' 迁移前代理 -> 200(经 w-a)')
// 顺便在用户工作区放个文件(迁移后要还在 —— 共享存储场景)
await json('/api/fs/upload', {
method: 'POST',
cookie: userCookie,
body: { path: 'proj', name: 'keep.txt', data: Buffer.from('survives migration').toString('base64') },
})
// ── 3) 迁移到 w-b ─────────────────────────────────────────────────────
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: 'rootpass123' } })
assert(adm.status === 200, 'admin login')
const adminCookie = adm.setCookie.split(';')[0]
r = await json('/api/admin/users/u1/dsh/migrate', {
method: 'POST',
cookie: adminCookie,
body: { targetHost: 'w-b' },
})
assert(r.status === 200, `迁移成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
assert(r.body.from === 'w-a' && r.body.to === 'w-b', `迁移方向 w-a→w-b(实际 ${JSON.stringify(r.body)})`)
assert(r.body.epoch === 2, `epoch 应 +1 到 2(实际 ${r.body.epoch})`)
row = await app.db.findUserInstance('u1', 'main')
assert(row.hostId === 'w-b' && row.epoch === 2, '归属已原子更新到 w-b / epoch=2')
assert((await a.instanceCount()) === 0, 'w-a 上实例已停(drain 生效)')
assert((await b.instanceCount()) === 1, 'w-b 上有 1 个实例')
console.log('② 迁移 -> w-a → w-b,host_id=w-b epoch=%d,源机实例已停', r.body.epoch)
// 迁移后代理照常(按新归属路由到 w-b)
proxyText = undefined
for (let i = 0; i < 20; i += 1) {
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie: userCookie } })
if (res.status === 200) {
proxyText = await res.text()
break
}
await sleep(100)
}
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), '迁移后代理 200(经 w-b)')
// ── 4) 数据还在(共享存储)────────────────────────────────────────────
const keptPath = join(sharedRoot, 'users', 'u1', 'ws', 'proj', 'keep.txt')
assert(existsSync(keptPath), `迁移后文件仍在:${keptPath}`)
const dl = await fetch(`${base}/api/fs/download?path=proj/keep.txt`, { headers: { cookie: userCookie } })
assert(dl.status === 200 && (await dl.text()) === 'survives migration', '迁移后仍能下载到原文')
console.log('③ 迁移后 -> 代理 200(经 w-b)、文件可读(数据不搬家)')
// ── 5) 已在该机 + 目标机不存在 ⇒ 明确报错(不静默)────────────────────
r = await json('/api/admin/users/u1/dsh/migrate', { method: 'POST', cookie: adminCookie, body: { targetHost: 'w-b' } })
assert(r.status === 409 && r.body.error === 'already_there', `重复迁移应 409 already_there(实际 ${r.status})`)
r = await json('/api/admin/users/u1/dsh/migrate', { method: 'POST', cookie: adminCookie, body: { targetHost: 'nope' } })
assert(r.status === 404 && r.body.error === 'unknown_host', `未知目标机应 404(实际 ${r.status})`)
console.log('④ 边界 -> already_there / unknown_host 都明确报错')
console.log('\nOK: 多 worker + 容量准入 + 迁移通过')
console.log(' ✓ 容量准入 ✓ 归属与实例一致 ✓ 迁移(drain→拉起→epoch+1) ✓ 迁移后代理/文件正常')
} finally {
// ⚠️ 收尾必须**停掉还活着的实例**:否则 fake-dsh 子进程会继承 stdout,
// 管道永不关闭 ⇒ ssh / CI 会一直挂在这里(2026-09-15 实测踩到)。
try {
await app?.supervisor?.stop('u1')
} catch {
/* best-effort */
}
await app?.close()
for (const h of agents) await h?.stop()
await sleep(500)
try {
rmSync(sharedRoot, { recursive: true, force: true })
} catch {
/* best-effort */
}
}
+214
View File
@@ -23,6 +23,9 @@ const HELP = `dshs — DSH server login orchestrator
Usage:
dshs [options] start the server
dshs bootstrap-admin [options] create the first admin
dshs worker [options] worker agent(cluster 模式:承载本机实例)
dshs doctor [--json] 单机自检(环境/隔离/存储/DB;非 0 退出 = 有硬失败)
dshs cluster status [--json] 全集群一屏(worker 目录 + 归属 + 过期租约)
Server options:
--port <n> Bind port (0 = ephemeral). Default 3080.
@@ -65,6 +68,138 @@ function toOverrides(values: ParsedValues): ConfigOverrides {
}
}
/**
* `dshs doctor`:**单机自检**(T08 S7;设计 §15.5)。
*
* 把"装机/排障要逐条手查"的东西固化成一条命令:环境 → 隔离能力 → 存储 → DB。
* **退出码非 0 = 有硬失败**(可直接用于 join 脚本的门禁);warn 不影响退出码。
*/
async function doctorCmd(args: string[]): Promise<void> {
const { values } = parseArgs({ args, options: { json: { type: 'boolean' } } })
const { execFileSync } = await import('node:child_process')
const { statfsSync, accessSync, constants } = await import('node:fs')
const config = resolveConfig({})
const lines: Array<{ level: 'ok' | 'warn' | 'fail'; item: string; detail: string }> = []
const add = (level: 'ok' | 'warn' | 'fail', item: string, detail: string): void => {
lines.push({ level, item, detail })
}
// ── 环境 ───────────────────────────────────────────────────────────────
add('ok', 'node', process.version)
let cgroup = 'unknown'
try {
cgroup = statfsSync('/sys/fs/cgroup').type === 0x63677270 ? 'v2' : 'v1'
} catch {
cgroup = 'unknown'
}
add(cgroup === 'unknown' ? 'warn' : 'ok', 'cgroup', cgroup)
let swap = ''
try {
// ⚠️ /proc/swaps 第一行是表头 ⇒ 要 NR>1,否则会把 "Filename" 当成设备名打出来
swap = execFileSync('/usr/bin/awk', ['NR>1 && NF>0 {print $1}', '/proc/swaps'], { encoding: 'utf8' })
.trim()
.replace(/\n+/g, ' ')
} catch {
swap = ''
}
add(swap === '' ? 'ok' : 'warn', 'swap', swap === '' ? '未启用' : `已启用(${swap})—— 实例超限会先换出而非被 OOM kill`)
for (const bin of ['bwrap', 'setpriv', 'systemd-run', 'nft', 'dsh']) {
const found = execFileSync('/usr/bin/which', [bin], { encoding: 'utf8' }).trim()
add(found === '' ? 'fail' : 'ok', bin, found === '' ? '缺失' : found)
}
let bwrapVersion = ''
try {
bwrapVersion = execFileSync('bwrap', ['--version'], { encoding: 'utf8' }).trim()
} catch {
bwrapVersion = ''
}
const minor = /bubblewrap (\d+)\.(\d+)/.exec(bwrapVersion)
if (minor !== null && Number(minor[2]) < 5) {
add('warn', 'bwrap 版本', `${bwrapVersion} —— **低于 0.5:不支持 --perms**(要改挂载点权限只能用 --tmpfs)`)
} else if (bwrapVersion !== '') {
add('ok', 'bwrap 版本', bwrapVersion)
}
// ── 隔离前提 ───────────────────────────────────────────────────────────
try {
const uid = execFileSync('setpriv', ['--reuid', '100001', '--regid', '100001', '--clear-groups', '--', 'id', '-u'], {
encoding: 'utf8',
}).trim()
add('ok', 'setpriv 降权', `可用(uid=${uid})`)
} catch {
add('fail', 'setpriv 降权', '失败(需要 root 或 CAP_SETUID)')
}
// ── 存储 ───────────────────────────────────────────────────────────────
try {
accessSync(config.dataRoot, constants.W_OK)
add('ok', 'dataRoot 可写', config.dataRoot)
} catch {
add('fail', 'dataRoot 可写', `${config.dataRoot} 不可写`)
}
// ── DB ─────────────────────────────────────────────────────────────────
try {
const { createDbAdapter } = await import('./db/index.js')
const db = await createDbAdapter(config)
const hosts = await db.listDshHosts()
await db.close()
add('ok', 'DB', `${config.dbUrl === undefined ? `sqlite ${config.dbPath}` : 'postgres'}(dsh_hosts ${hosts.length} 条)`)
} catch (err) {
add('fail', 'DB', err instanceof Error ? err.message : String(err))
}
const failures = lines.filter((l) => l.level === 'fail').length
if (values.json === true) {
process.stdout.write(JSON.stringify({ ok: failures === 0, checks: lines }, null, 2) + '\n')
} else {
for (const l of lines) {
const mark = l.level === 'ok' ? '✓' : l.level === 'warn' ? '!' : '✗'
process.stdout.write(`${mark} ${l.item.padEnd(16)} ${l.detail}\n`)
}
process.stdout.write(`\n${failures === 0 ? 'OK:无硬失败' : `${failures} 项硬失败`}\n`)
}
if (failures > 0) process.exit(2)
}
/**
* `dshs cluster status`:**全集群一屏**(T08 S7;设计 §15.5)。
* 读的是**控制面 DB**(Manager 侧运行),输出 worker 目录 + 实例归属 + 过期租约。
*/
async function clusterStatusCmd(args: string[]): Promise<void> {
const { values } = parseArgs({ args, options: { json: { type: 'boolean' } } })
const config = resolveConfig({})
const { createDbAdapter } = await import('./db/index.js')
const db = await createDbAdapter(config)
try {
const [hosts, expired] = await Promise.all([
db.listDshHosts(),
db.listExpiredInstanceLeases(Date.now()),
])
const byHost = new Map<string, number>()
for (const h of hosts) byHost.set(h.id, (await db.listInstancesByHost(h.id)).length)
if (values.json === true) {
process.stdout.write(JSON.stringify({ deployMode: config.deployMode, hosts, expired }, null, 2) + '\n')
return
}
process.stdout.write(`deployMode=${config.deployMode} db=${config.dbUrl === undefined ? config.dbPath : 'postgres'}\n\n`)
process.stdout.write('WORKER 状态 容量(MB) 已用 实例 最后心跳\n')
for (const h of hosts) {
const hb = h.lastHeartbeat === null ? '从未' : new Date(h.lastHeartbeat).toISOString().replace('T', ' ').slice(0, 19)
process.stdout.write(
`${h.id.padEnd(26)} ${h.status.padEnd(8)} ${String(h.capacityMb).padStart(8)} ${String(h.usedMb).padStart(6)} ` +
`${String(byHost.get(h.id) ?? 0).padStart(6)} ${hb}\n`,
)
}
process.stdout.write(`\n租约已过期(需人工确认后才可接管,见 R9):${expired.length} 个\n`)
for (const inst of expired) {
process.stdout.write(` ${inst.userId} host=${inst.hostId ?? '-'} epoch=${inst.epoch} 过期于 ${new Date(inst.leaseUntil).toISOString()}\n`)
}
} finally {
await db.close()
}
}
async function bootstrapAdmin(args: string[]): Promise<void> {
const { values } = parseArgs({
args,
@@ -140,6 +275,69 @@ async function runServer(args: string[]): Promise<void> {
process.on('SIGTERM', () => void shutdown('SIGTERM'))
}
/**
* 运行 **worker agent**(T08 S3;设计 §11.2)。
*
* 它是 Worker 上唯一的"被拨入口":把本机的实例生命周期(launch/stop/status/endpoint/fence)
* 暴露给 Manager。**不连控制面 DB** —— 凭据(apiKey)与 uid 由 Manager 在 launch 时投递,
* 只存内存(与 k8s 用 per-user Secret 同一思路)。**不是"不许有数据库"**:插件业务数据在
* 实例 home 里、由实例自己读写(设计 §1.3 数据分层)。
*/
async function runWorker(args: string[]): Promise<void> {
const { values } = parseArgs({
args,
options: {
port: { type: 'string' },
host: { type: 'string' },
'host-id': { type: 'string' },
token: { type: 'string' },
'instance-host': { type: 'string' },
'log-level': { type: 'string' },
help: { type: 'boolean', short: 'h' },
},
})
if (values.help === true) {
process.stdout.write(
'usage: dshs worker --token <secret> [--port 9000] [--host 0.0.0.0]\n' +
' [--host-id <id>] [--instance-host <addr>]\n' +
' 密钥也可用 DSHS_CLUSTER_AGENT_TOKEN;host-id 默认取主机名。\n',
)
return
}
const token =
(typeof values.token === 'string' ? values.token : undefined) ?? process.env.DSHS_CLUSTER_AGENT_TOKEN
if (token === undefined || token === '') {
console.error('worker requires --token or DSHS_CLUSTER_AGENT_TOKEN')
process.exit(2)
}
const config = resolveConfig({
logLevel: typeof values['log-level'] === 'string' ? values['log-level'] : undefined,
clusterHostId: typeof values['host-id'] === 'string' ? values['host-id'] : undefined,
clusterInstanceHost: typeof values['instance-host'] === 'string' ? values['instance-host'] : undefined,
})
const { buildWorkerAgent } = await import('./worker/agent.js')
const host = typeof values.host === 'string' ? values.host : '0.0.0.0'
const port = Number(typeof values.port === 'string' ? values.port : 9000)
const agent = buildWorkerAgent(config, {
hostId: config.clusterHostId,
token,
port,
host,
instanceHost: config.clusterInstanceHost,
logLevel: config.logLevel,
})
await agent.app.listen({ host, port })
agent.app.log.info(`worker agent listening on http://${host}:${port} (hostId ${config.clusterHostId})`)
const shutdown = async (signal: string): Promise<void> => {
agent.app.log.info(`received ${signal}, shutting down`)
await agent.stop()
process.exit(0)
}
process.on('SIGINT', () => void shutdown('SIGINT'))
process.on('SIGTERM', () => void shutdown('SIGTERM'))
}
async function uidForUserCmd(args: string[]): Promise<void> {
const { values, positionals } = parseArgs({
args,
@@ -166,6 +364,22 @@ async function main(): Promise<void> {
await bootstrapAdmin(rest)
return
}
if (first === 'worker') {
await runWorker(rest)
return
}
if (first === 'doctor') {
await doctorCmd(rest)
return
}
if (first === 'cluster') {
if (rest[0] === 'status') {
await clusterStatusCmd(rest.slice(1))
return
}
process.stderr.write('usage: dshs cluster status [--json]\n')
process.exit(2)
}
if (first === 'uid-for-user') {
await uidForUserCmd(rest)
return
+28 -3
View File
@@ -15,7 +15,7 @@ export type IsolationMode = 'soft' | 'account'
/** Deployment mode. `local` = single-host child_process (setuid/iptables);
* `k8s` = multi-replica control plane spawning per-user DSH Pods via the K8s API. */
export type DeployMode = 'local' | 'k8s'
export type DeployMode = 'local' | 'k8s' | 'cluster'
/** Resolved, immutable runtime configuration. */
export interface ServerConfig {
@@ -96,6 +96,21 @@ export interface ServerConfig {
k8sServiceAccount: string
/** This replica's identity for leader election (POD_NAME, else hostname). */
podName: string
// ── cluster 模式(T08 S3/S4;设计 §1.1)────────────────────────────────
/** 本机在 `dsh_hosts.id` 里的标识(`deployMode=cluster` 时必填语义)。 */
clusterHostId: string
/** 本机 worker agent 的**基址**(Manager 侧用它投递实例操作),如 `http://127.0.0.1:9000`。 */
clusterAgentUrl: string
/** 与 agent 约定的共享密钥(仅内网 + nft 白名单)。 */
clusterAgentToken: string
/** agent 返回给 Manager 做代理的实例地址(同机 1a = `127.0.0.1`)。 */
clusterInstanceHost: string
/**
* **worker 上**的 dataRoot(T08 S5)。
* 空 = 与本地 `dataRoot` 相同(1a 形态)。多机部署必须显式配置 —— 而且是**基线约定**:
* 所有 worker 的 dataRoot 必须是同一个绝对路径(同镜像即可满足,设计 §14.3)。
*/
clusterWorkerDataRoot: string
}
/** Untyped overrides collected from argv / env. */
@@ -137,6 +152,11 @@ export interface ConfigOverrides {
egressCidrs?: string[]
k8sServiceAccount?: string
podName?: string
clusterHostId?: string
clusterAgentUrl?: string
clusterAgentToken?: string
clusterInstanceHost?: string
clusterWorkerDataRoot?: string
}
const DEFAULT_HOST = '127.0.0.1'
@@ -223,8 +243,8 @@ function toIsolationMode(value: string | undefined): IsolationMode | undefined {
function toDeployMode(value: string | undefined): DeployMode | undefined {
if (value === undefined) return undefined
const normalized = value.trim().toLowerCase()
if (normalized === 'local' || normalized === 'k8s') return normalized
throw new Error(`invalid deploy mode "${value}" (expected "local" or "k8s")`)
if (normalized === 'local' || normalized === 'k8s' || normalized === 'cluster') return normalized
throw new Error(`invalid deploy mode "${value}" (expected "local", "k8s" or "cluster")`)
}
/**
@@ -327,5 +347,10 @@ export function resolveConfig(overrides: ConfigOverrides = {}): ServerConfig {
k8sServiceAccount:
overrides.k8sServiceAccount ?? process.env.DSHS_K8S_SERVICE_ACCOUNT ?? DEFAULT_K8S_SERVICE_ACCOUNT,
podName: overrides.podName ?? process.env.POD_NAME ?? hostname(),
clusterHostId: overrides.clusterHostId ?? process.env.DSHS_CLUSTER_HOST_ID ?? hostname(),
clusterAgentUrl: overrides.clusterAgentUrl ?? process.env.DSHS_CLUSTER_AGENT_URL ?? '',
clusterAgentToken: overrides.clusterAgentToken ?? process.env.DSHS_CLUSTER_AGENT_TOKEN ?? '',
clusterInstanceHost: overrides.clusterInstanceHost ?? process.env.DSHS_CLUSTER_INSTANCE_HOST ?? '127.0.0.1',
clusterWorkerDataRoot: overrides.clusterWorkerDataRoot ?? process.env.DSHS_CLUSTER_WORKER_DATA_ROOT ?? '',
}
}
+39
View File
@@ -7,12 +7,15 @@
import type {
BusinessPlugin,
ClaimResult,
CredentialKey,
CredentialKeyMeta,
CredentialLandingRow,
CreateSessionInput,
CreateUserInput,
Domain,
DshHost,
DshHostStatus,
DshInstance,
DshInstanceRole,
DshInstanceStatus,
@@ -20,6 +23,7 @@ import type {
SessionRow,
SessionUser,
UpsertBusinessPluginInput,
UpsertDshHostInput,
UpsertDshInstanceInput,
User,
UserRole,
@@ -104,6 +108,41 @@ export interface DbAdapter {
): Promise<boolean>
deleteInstance(id: string): Promise<boolean>
deleteUserInstances(userId: string): Promise<void>
// ── 集群化:worker 注册表 + 实例归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)──
// ⚠️ local 模式**不写**这些表(`LocalSpawner` 靠进程内 Map + 单机互斥),
// 所以这些方法在单机路径上恒为"空/未认领",不影响现有行为。
/** 注册/更新一台 worker(join 幂等:同 id 重复执行 = 更新)。 */
upsertDshHost(input: UpsertDshHostInput): Promise<DshHost>
findDshHost(id: string): Promise<DshHost | undefined>
listDshHosts(): Promise<DshHost[]>
/** 心跳/状态上报:可只改状态,或同时带上容量水位与心跳时间。 */
setDshHostStatus(
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): Promise<boolean>
/**
* **原子抢占**某用户的 main 实例归属(承重墙,见设计 §3.2)。
* 仅当"无人持有 **或** 租约已过期"才成功;成功时 `epoch` +1(fencing)。
* 返回 `ok:false` = 有人在管 ⇒ 调用方**退让**(不是接管)。
*/
claimInstance(
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): Promise<ClaimResult>
/** 续租。**必须带 epoch**:不匹配说明已被他人抢占 ⇒ 本次续租失败(fencing)。 */
renewInstanceLease(userId: string, hostId: string, epoch: number, ttlMs: number): Promise<boolean>
/** 主动释放(停实例时)。同样带 epoch 校验,避免误清他人的归属。 */
releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean>
/** 钉住归属(首次触达工作区时用):只写 `host_id`,不动 epoch/租约。 */
pinInstanceHost(userId: string, hostId: string): Promise<void>
/** 租约已过期、但仍标着归属的实例 —— 供巡检/自愈(**不代表可以立即接管**,见 R9)。 */
listExpiredInstanceLeases(now: number): Promise<DshInstance[]>
/** 某 worker 上的全部实例 —— 对账用**一次拿回整机**(替代逐用户查询)。 */
listInstancesByHost(hostId: string): Promise<DshInstance[]>
// lifecycle
close(): Promise<void>
}
+156 -1
View File
@@ -11,20 +11,25 @@ import type { DbAdapter } from './adapter.js'
import { mapPgError } from './errors.js'
import { runPgMigrations } from './schema.js'
import {
clusterInstanceId,
toBusinessPlugin,
toDomain,
toDshHost,
toDshInstance,
toPublicUser,
toSession,
toUser,
toWorkspace,
type BusinessPlugin,
type ClaimResult,
type CredentialKey,
type CredentialKeyMeta,
type CredentialLandingRow,
type CreateSessionInput,
type CreateUserInput,
type Domain,
type DshHost,
type DshHostStatus,
type DshInstance,
type DshInstanceRole,
type DshInstanceStatus,
@@ -32,6 +37,7 @@ import {
type SessionRow,
type SessionUser,
type UpsertBusinessPluginInput,
type UpsertDshHostInput,
type UpsertDshInstanceInput,
type User,
type UserRole,
@@ -47,8 +53,10 @@ types.setTypeParser(20, (value: string) => Number(value))
const USER_COLS = 'id, username, pass_hash, role, home_dir, api_key_ref, created_at, approved_by, uid'
const DOMAIN_COLS = 'id, user_id, domain, verified, nginx_config, updated_at'
const BUSINESS_PLUGIN_COLS = 'id, name, description, version, tgz_path, file_size, uploaded_by, created_at, updated_at'
const HOST_COLS = 'id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat'
const INSTANCE_COLS =
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch'
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch, '
+ 'host_id, epoch, heartbeat_at, lease_until' // v7 集群化归属/租约(T08 S2)—— 漏了它们会让 hostId 恒为 null
/** Run `fn` on a dedicated client inside a BEGIN/COMMIT/ROLLBACK transaction. */
export async function withTx<T>(pool: Pool, fn: (client: PoolClient) => Promise<T>): Promise<T> {
@@ -587,6 +595,153 @@ export class PgAdapter implements DbAdapter {
await this.pool.query('DELETE FROM dsh_instances WHERE user_id = $1', [userId])
}
// ── 集群化:worker 注册表 + 归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)────────
// 与 `repo.ts` 的同名 SQLite 实现**逐条对齐**(两套实现并存是本库既有事实,
// 见档案 19 §C8):任何 schema/语义变更都要**两侧同改**,否则切库时才炸。
async upsertDshHost(input: UpsertDshHostInput): Promise<DshHost> {
try {
const { rows } = await this.pool.query(
`INSERT INTO dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat)
VALUES ($1, $2, $3, $4, 0, $5, NULL)
ON CONFLICT(id) DO UPDATE SET
endpoint = excluded.endpoint,
agent_token = excluded.agent_token,
capacity_mb = excluded.capacity_mb,
status = excluded.status
RETURNING id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat`,
[input.id, input.endpoint, input.agentToken, input.capacityMb, input.status ?? 'up'],
)
return toDshHost(rows[0] as Record<string, unknown>)
} catch (e) {
mapPgError(e)
}
}
async findDshHost(id: string): Promise<DshHost | undefined> {
const { rows } = await this.pool.query(
`SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = $1`,
[id],
)
return rows.length > 0 ? toDshHost(rows[0] as Record<string, unknown>) : undefined
}
async listDshHosts(): Promise<DshHost[]> {
const { rows } = await this.pool.query(`SELECT ${HOST_COLS} FROM dsh_hosts ORDER BY id ASC`)
return rows.map((row) => toDshHost(row as Record<string, unknown>))
}
async setDshHostStatus(
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): Promise<boolean> {
const result = await this.pool.query(
`UPDATE dsh_hosts
SET status = $1,
used_mb = COALESCE($2, used_mb),
last_heartbeat = COALESCE($3, last_heartbeat)
WHERE id = $4`,
[status, usedMb ?? null, heartbeatAt ?? null, id],
)
return (result.rowCount ?? 0) > 0
}
/**
* **原子抢占**(承重墙):PG 侧用 `UPDATE … RETURNING` —— 只有真正更新到行才返回行,
* 比"先读后写"少一次竞态窗口(SQLite 侧用 `changes` 判定,语义等价)。
*/
async claimInstance(
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): Promise<ClaimResult> {
const now = Date.now()
const id = clusterInstanceId(userId)
await this.pool.query(
`INSERT INTO dsh_instances (id, user_id, role, status) VALUES ($1, $2, 'main', 'starting')
ON CONFLICT(id) DO NOTHING`,
[id, userId],
)
// folder/patch 一起落库:迁移要能复现启动参数(见 repo.ts 同名处注释)
const res = await this.pool.query(
`UPDATE dsh_instances
SET host_id = $1, epoch = epoch + 1, heartbeat_at = $2, lease_until = $3,
folder = COALESCE($4, folder), patch = COALESCE($5, patch)
WHERE id = $6 AND (host_id IS NULL OR lease_until < $2)
RETURNING epoch, lease_until`,
[hostId, now, now + ttlMs, meta?.folder ?? null, meta?.patch ?? null, id],
)
if (res.rows.length > 0) {
const row = res.rows[0] as { epoch: number; lease_until: number }
return { ok: true, epoch: row.epoch, leaseUntil: row.lease_until }
}
const cur = await this.pool.query('SELECT host_id, lease_until FROM dsh_instances WHERE id = $1', [id])
const row = cur.rows[0] as { host_id: string | null; lease_until: number } | undefined
return { ok: false, holder: row?.host_id ?? null, leaseUntil: row?.lease_until ?? 0 }
}
async renewInstanceLease(
userId: string,
hostId: string,
epoch: number,
ttlMs: number,
): Promise<boolean> {
const now = Date.now()
const result = await this.pool.query(
`UPDATE dsh_instances SET heartbeat_at = $1, lease_until = $2
WHERE id = $3 AND host_id = $4 AND epoch = $5`,
[now, now + ttlMs, clusterInstanceId(userId), hostId, epoch],
)
return (result.rowCount ?? 0) > 0
}
/** 只清租约、**保留 host_id**(见 repo.ts 同名函数的长注释)。 */
async releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean> {
const result = await this.pool.query(
`UPDATE dsh_instances SET lease_until = 0
WHERE id = $1 AND host_id = $2 AND epoch = $3`,
[clusterInstanceId(userId), hostId, epoch],
)
return (result.rowCount ?? 0) > 0
}
/** 钉住归属(首次触达工作区时用):只写 host_id。 */
async pinInstanceHost(userId: string, hostId: string): Promise<void> {
const now = Date.now()
const id = clusterInstanceId(userId)
await this.pool.query(
`INSERT INTO dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
VALUES ($1, $2, 'main', 'stopped', $3, 0, 0, 0)
ON CONFLICT(id) DO NOTHING`,
[id, userId, hostId],
)
await this.pool.query(
`UPDATE dsh_instances SET host_id = $1 WHERE id = $2 AND (host_id IS NULL OR lease_until < $3)`,
[hostId, id, now],
)
}
async listExpiredInstanceLeases(now: number): Promise<DshInstance[]> {
const { rows } = await this.pool.query(
`SELECT ${INSTANCE_COLS} FROM dsh_instances
WHERE role = 'main' AND host_id IS NOT NULL AND lease_until < $1 AND status <> 'stopped'
ORDER BY lease_until ASC`,
[now],
)
return rows.map((row) => toDshInstance(row as Record<string, unknown>))
}
async listInstancesByHost(hostId: string): Promise<DshInstance[]> {
const { rows } = await this.pool.query(
`SELECT ${INSTANCE_COLS} FROM dsh_instances WHERE host_id = $1 ORDER BY user_id ASC`,
[hostId],
)
return rows.map((row) => toDshInstance(row as Record<string, unknown>))
}
async close(): Promise<void> {
await this.pool.end()
}
+168 -1
View File
@@ -12,20 +12,25 @@ import { randomUUID } from 'node:crypto'
import type { Database } from './connection.js'
import { prepare } from './prepared.js'
import {
clusterInstanceId,
toBusinessPlugin,
toDomain,
toDshHost,
toDshInstance,
toPublicUser,
toSession,
toUser,
toWorkspace,
type BusinessPlugin,
type ClaimResult,
type CredentialKey,
type CredentialKeyMeta,
type CredentialLandingRow,
type CreateSessionInput,
type CreateUserInput,
type Domain,
type DshHost,
type DshHostStatus,
type DshInstance,
type DshInstanceRole,
type DshInstanceStatus,
@@ -33,6 +38,7 @@ import {
type SessionRow,
type SessionUser,
type UpsertBusinessPluginInput,
type UpsertDshHostInput,
type UpsertDshInstanceInput,
type User,
type UserRole,
@@ -43,7 +49,8 @@ const USER_COLS = 'id, username, pass_hash, role, home_dir, api_key_ref, created
const DOMAIN_COLS = 'id, user_id, domain, verified, nginx_config, updated_at'
const BUSINESS_PLUGIN_COLS = 'id, name, description, version, tgz_path, file_size, uploaded_by, created_at, updated_at'
const INSTANCE_COLS =
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch'
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch, '
+ 'host_id, epoch, heartbeat_at, lease_until' // v7 集群化归属/租约(T08 S2)—— 漏了它们会让 hostId 恒为 null
export function createUser(db: Database, input: CreateUserInput, baseUid: number): User {
const createdAt = Date.now()
@@ -582,3 +589,163 @@ export function deleteBusinessPlugin(db: Database, id: string): boolean {
const info = prepare(db, 'DELETE FROM business_plugins WHERE id = ?').run(id)
return info.changes > 0
}
// ── 集群化:worker 注册表 + 实例归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)────────
//
// ⚠️ local 模式**不调用**这些函数(`LocalSpawner` 靠进程内 Map + 单机互斥),
// 所以它们的存在不会改变现有单机行为。
const HOST_COLS = 'id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat'
/** 注册/更新一台 worker。join 幂等:同 id 重复执行 = 更新(并把它标回 `up`)。 */
export function upsertDshHost(db: Database, input: UpsertDshHostInput): DshHost {
prepare(db, `
INSERT INTO dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat)
VALUES (?, ?, ?, ?, 0, ?, NULL)
ON CONFLICT(id) DO UPDATE SET
endpoint = excluded.endpoint,
agent_token = excluded.agent_token,
capacity_mb = excluded.capacity_mb,
status = excluded.status
`).run(input.id, input.endpoint, input.agentToken, input.capacityMb, input.status ?? 'up')
const row = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = ?`).get(input.id)
return toDshHost(row as Record<string, unknown>)
}
export function findDshHost(db: Database, id: string): DshHost | undefined {
const row = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = ?`).get(id)
return row ? toDshHost(row as Record<string, unknown>) : undefined
}
export function listDshHosts(db: Database): DshHost[] {
const rows = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts ORDER BY id ASC`).all() as Array<
Record<string, unknown>
>
return rows.map((row) => toDshHost(row))
}
/** 心跳/状态上报(只更新显式给出的字段,避免 heartbeat 覆盖 status)。 */
export function setDshHostStatus(
db: Database,
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): boolean {
const info = prepare(db, `
UPDATE dsh_hosts
SET status = ?,
used_mb = COALESCE(?, used_mb),
last_heartbeat = COALESCE(?, last_heartbeat)
WHERE id = ?
`).run(status, usedMb ?? null, heartbeatAt ?? null, id)
return info.changes > 0
}
/**
* **原子抢占**某用户 main 实例的归属(承重墙)。仅当"无人持有 **或** 租约已过期"才成功,
* 成功时 `epoch` +1(fencing token)。失败时返回当前持有者与租约到期时刻。
*/
export function claimInstance(
db: Database,
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): ClaimResult {
const now = Date.now()
const id = clusterInstanceId(userId)
// 新用户没有 dsh_instances 行 ⇒ 先保证行存在(否则 UPDATE 影响 0 行被误判为"有人在管")。
prepare(db, `
INSERT INTO dsh_instances (id, user_id, role, status)
VALUES (?, ?, 'main', 'starting')
ON CONFLICT(id) DO NOTHING
`).run(id, userId)
// ⚠️ **必须把 folder/patch 一起落库**(2026-09-15 实测踩到):集群模式下实例行是这里建的,
// 而 local 模式不写库 ⇒ 若这里不记,`folder` 永远是 NULL,**迁移时复现不了启动参数**
// (表现为 `bwrap: Can't chdir to :` 空路径 ⇒ 崩溃循环)。用 COALESCE 保证不覆盖已有值。
const info = prepare(db, `
UPDATE dsh_instances
SET host_id = ?, epoch = epoch + 1, heartbeat_at = ?, lease_until = ?,
folder = COALESCE(?, folder), patch = COALESCE(?, patch)
WHERE id = ? AND (host_id IS NULL OR lease_until < ?)
`).run(hostId, now, now + ttlMs, meta?.folder ?? null, meta?.patch ?? null, id, now)
const row = prepare(db, 'SELECT host_id, epoch, lease_until FROM dsh_instances WHERE id = ?').get(id) as
| { host_id: string | null; epoch: number; lease_until: number }
| undefined
if (row === undefined) return { ok: false, holder: null, leaseUntil: 0 }
return info.changes > 0
? { ok: true, epoch: row.epoch, leaseUntil: row.lease_until }
: { ok: false, holder: row.host_id, leaseUntil: row.lease_until }
}
/** 续租。**必须带 epoch**:不匹配说明已被他人抢占 ⇒ 返回 false(fencing 生效)。 */
export function renewInstanceLease(
db: Database,
userId: string,
hostId: string,
epoch: number,
ttlMs: number,
): boolean {
const now = Date.now()
const info = prepare(db, `
UPDATE dsh_instances SET heartbeat_at = ?, lease_until = ?
WHERE id = ? AND host_id = ? AND epoch = ?
`).run(now, now + ttlMs, clusterInstanceId(userId), hostId, epoch)
return info.changes > 0
}
/**
* 主动释放**租约**(停实例时)。带 epoch 校验,避免误清他人的归属。
*
* ⚠️ **只清 `lease_until`,保留 `host_id`**(2026-09-15 生产切换暴露):
* `host_id` 的语义是「**这个用户的数据在哪台机器**」—— 用户的工作区在**本地盘**上,
* 把归属一起清掉就等于**丢掉粘性锚点**,下次启动可能被调度到没有他数据的机器上(工作区看起来是空的)。
* 「谁现在在托管」是**租约**(`lease_until`)的语义,所以释放只该清租约。
*/
export function releaseInstanceLease(db: Database, userId: string, hostId: string, epoch: number): boolean {
const info = prepare(db, `
UPDATE dsh_instances SET lease_until = 0
WHERE id = ? AND host_id = ? AND epoch = ?
`).run(clusterInstanceId(userId), hostId, epoch)
return info.changes > 0
}
/**
* **钉住**某用户的归属(首次触达其工作区时用):只写 `host_id`,不动 epoch/租约。
*
* 为什么需要:新用户还没有归属,`selectHost` 会在**写文件那一步**与**launch 那一步**各自选一次,
* 两次可能选到不同机器 ⇒ 「文件写到 A、实例起在 B」⇒ 实例看不到自己的文件(2026-09-15 实测)。
* 首次触达就把归属钉住,后续(含 launch)都走粘性,两面必然一致。
*/
export function pinInstanceHost(db: Database, userId: string, hostId: string): void {
const now = Date.now()
const id = clusterInstanceId(userId)
prepare(db, `
INSERT INTO dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
VALUES (?, ?, 'main', 'stopped', ?, 0, 0, 0)
ON CONFLICT(id) DO NOTHING
`).run(id, userId, hostId)
prepare(db, `
UPDATE dsh_instances SET host_id = ?
WHERE id = ? AND (host_id IS NULL OR lease_until < ?)
`).run(hostId, id, now)
}
/** 租约过期但仍标着归属的 main 实例(供巡检/自愈;**不等于可以立即接管**,见 R9)。 */
export function listExpiredInstanceLeases(db: Database, now: number): DshInstance[] {
const rows = prepare(db, `
SELECT ${INSTANCE_COLS} FROM dsh_instances
WHERE role = 'main' AND host_id IS NOT NULL AND lease_until < ? AND status <> 'stopped'
ORDER BY lease_until ASC
`).all(now) as Array<Record<string, unknown>>
return rows.map((row) => toDshInstance(row))
}
/** 某 worker 上的全部实例 —— 对账**一次拿回整机**(替代逐用户查询)。 */
export function listInstancesByHost(db: Database, hostId: string): DshInstance[] {
const rows = prepare(db, `SELECT ${INSTANCE_COLS} FROM dsh_instances WHERE host_id = ? ORDER BY user_id ASC`).all(
hostId,
) as Array<Record<string, unknown>>
return rows.map((row) => toDshInstance(row))
}
+54
View File
@@ -292,6 +292,59 @@ ALTER TABLE credential_vault ADD COLUMN models TEXT;
ALTER TABLE users ADD COLUMN shared_model_enabled INTEGER NOT NULL DEFAULT 1;
`
// v7: 集群化 —— worker 注册表 + 实例归属/租约(T08 S2;设计 §3.1/§3.2)。
//
// 为什么需要它:local 模式靠"进程内 Map + 单机"天然保证「一个用户只有一个活实例」;
// 多机后这个保证必须落到 DB 的**原子 CAS** 上,否则两个 worker 会同时写同一个
// `$DSH_HOME`(会话日志 append 冲突 ⇒ 数据损坏)。
//
// · `dsh_hosts` = worker 注册表:agent 地址、容量、水位、心跳时间。
// · `dsh_instances.{host_id, epoch, heartbeat_at, lease_until}` = 归属与租约。
// `epoch` 是 **fencing token**:抢占时 +1,旧持有者的写入据此被拒(防脑裂双写)。
//
// 抢占语义(两方言同款,见 `repo.ts` 的 claimInstance / `pg.ts` 同名方法):
// `INSERT … ON CONFLICT(id) DO UPDATE SET … WHERE host_id IS NULL OR lease_until < now`
// —— 冲突时仅在"无人持有或租约过期"才更新;否则**不动行也不报错**,
// 调用方以「受影响行数 0」判定"有人在管"。
// ⚠️ 时间戳一律 **epoch 毫秒 BIGINT**(与全库一致,勿用 timestamptz)。
const SQLITE_V7 = `
CREATE TABLE IF NOT EXISTS dsh_hosts (
id TEXT PRIMARY KEY,
endpoint TEXT NOT NULL,
agent_token TEXT NOT NULL,
capacity_mb INTEGER NOT NULL DEFAULT 0,
used_mb INTEGER NOT NULL DEFAULT 0,
status TEXT NOT NULL DEFAULT 'up'
CHECK (status IN ('up','draining','down')),
last_heartbeat INTEGER
);
ALTER TABLE dsh_instances ADD COLUMN host_id TEXT;
ALTER TABLE dsh_instances ADD COLUMN epoch INTEGER NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN heartbeat_at INTEGER NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN lease_until INTEGER NOT NULL DEFAULT 0;
CREATE INDEX IF NOT EXISTS idx_dsh_instances_host ON dsh_instances (host_id);
CREATE INDEX IF NOT EXISTS idx_dsh_instances_lease ON dsh_instances (lease_until);
`
const PG_V7 = `
CREATE TABLE IF NOT EXISTS dsh_hosts (
id TEXT PRIMARY KEY,
endpoint TEXT NOT NULL,
agent_token TEXT NOT NULL,
capacity_mb BIGINT NOT NULL DEFAULT 0,
used_mb BIGINT NOT NULL DEFAULT 0,
status TEXT NOT NULL DEFAULT 'up'
CHECK (status IN ('up','draining','down')),
last_heartbeat BIGINT
);
ALTER TABLE dsh_instances ADD COLUMN host_id TEXT;
ALTER TABLE dsh_instances ADD COLUMN epoch BIGINT NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN heartbeat_at BIGINT NOT NULL DEFAULT 0;
ALTER TABLE dsh_instances ADD COLUMN lease_until BIGINT NOT NULL DEFAULT 0;
CREATE INDEX IF NOT EXISTS idx_dsh_instances_host ON dsh_instances (host_id);
CREATE INDEX IF NOT EXISTS idx_dsh_instances_lease ON dsh_instances (lease_until);
`
interface Migration {
version: number
name: string
@@ -306,6 +359,7 @@ const MIGRATIONS: readonly Migration[] = [
{ version: 4, name: 'instance desired state', sqlite: SQLITE_V4, pg: PG_V4 },
{ version: 5, name: 'business plugin candidate pool', sqlite: SQLITE_V5, pg: PG_V5 },
{ version: 6, name: 'user model providers', sqlite: SQLITE_V6, pg: PG_V6 },
{ version: 7, name: 'cluster host registry + instance lease', sqlite: SQLITE_V7, pg: PG_V7 },
]
/** Apply unapplied SQLite migrations inside a single transaction. */
+73
View File
@@ -57,18 +57,32 @@ import {
setUserRole as setUserRoleSync,
setUserUid as setUserUidSync,
toggleCredentialKey as toggleCredentialKeySync,
// 集群化(v7;T08 S2)
claimInstance as claimInstanceSync,
findDshHost as findDshHostSync,
listDshHosts as listDshHostsSync,
listExpiredInstanceLeases as listExpiredInstanceLeasesSync,
listInstancesByHost as listInstancesByHostSync,
pinInstanceHost as pinInstanceHostSync,
releaseInstanceLease as releaseInstanceLeaseSync,
renewInstanceLease as renewInstanceLeaseSync,
setDshHostStatus as setDshHostStatusSync,
upsertDshHost as upsertDshHostSync,
upsertBusinessPlugin as upsertBusinessPluginSync,
upsertDomain as upsertDomainSync,
upsertInstance as upsertInstanceSync,
} from './repo.js'
import type {
BusinessPlugin,
ClaimResult,
CredentialKey,
CredentialKeyMeta,
CredentialLandingRow,
CreateSessionInput,
CreateUserInput,
Domain,
DshHost,
DshHostStatus,
DshInstance,
DshInstanceRole,
DshInstanceStatus,
@@ -76,6 +90,7 @@ import type {
SessionRow,
SessionUser,
UpsertBusinessPluginInput,
UpsertDshHostInput,
UpsertDshInstanceInput,
User,
UserRole,
@@ -321,6 +336,64 @@ export class SqliteAdapter implements DbAdapter {
deleteUserInstancesSync(this.db, userId)
}
// ── 集群化:worker 注册表 + 归属/租约(v7;T08 S2)────────────────────────
// local 模式不会走到这些方法(`LocalSpawner` 不写库),它们只是让
// **SQLite 侧与 PG 侧行为一致** —— 测试与单机试跑都需要。
async upsertDshHost(input: UpsertDshHostInput): Promise<DshHost> {
try {
return upsertDshHostSync(this.db, input)
} catch (e) {
mapSqliteError(e)
}
}
async findDshHost(id: string): Promise<DshHost | undefined> {
return findDshHostSync(this.db, id)
}
async listDshHosts(): Promise<DshHost[]> {
return listDshHostsSync(this.db)
}
async setDshHostStatus(
id: string,
status: DshHostStatus,
usedMb?: number,
heartbeatAt?: number,
): Promise<boolean> {
return setDshHostStatusSync(this.db, id, status, usedMb, heartbeatAt)
}
async claimInstance(
userId: string,
hostId: string,
ttlMs: number,
meta?: { folder?: string; patch?: string },
): Promise<ClaimResult> {
return claimInstanceSync(this.db, userId, hostId, ttlMs, meta)
}
async renewInstanceLease(userId: string, hostId: string, epoch: number, ttlMs: number): Promise<boolean> {
return renewInstanceLeaseSync(this.db, userId, hostId, epoch, ttlMs)
}
async releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean> {
return releaseInstanceLeaseSync(this.db, userId, hostId, epoch)
}
async pinInstanceHost(userId: string, hostId: string): Promise<void> {
pinInstanceHostSync(this.db, userId, hostId)
}
async listExpiredInstanceLeases(now: number): Promise<DshInstance[]> {
return listExpiredInstanceLeasesSync(this.db, now)
}
async listInstancesByHost(hostId: string): Promise<DshInstance[]> {
return listInstancesByHostSync(this.db, hostId)
}
async close(): Promise<void> {
this.db.close()
}
+74
View File
@@ -90,6 +90,15 @@ export interface DshInstance {
folder: string | null
/** Rendered Cordis patch content (not a path — the control plane holds no user volume). */
patch: string | null
// ── 集群化归属与租约(v7;T08 S2 / 设计 §3.1)—— local 模式下恒为 null/0 ──
/** 托管该实例的 worker(`dsh_hosts.id`);null = 未被任何 worker 认领。 */
hostId: string | null
/** **fencing token**:每次抢占 +1;旧持有者的写入据此被拒(防脑裂双写)。 */
epoch: number
/** 最近一次心跳(epoch 毫秒)。 */
heartbeatAt: number
/** 租约到期时刻(epoch 毫秒);早于 now 即可被他人抢占。 */
leaseUntil: number
}
/** A named per-user credential key (secret never exposed). */
@@ -262,6 +271,10 @@ export function toDshInstance(row: Record<string, unknown>): DshInstance {
lastError: (row.last_error as string | null) ?? null,
folder: (row.folder as string | null) ?? null,
patch: (row.patch as string | null) ?? null,
hostId: (row.host_id as string | null) ?? null,
epoch: (row.epoch as number | null) ?? 0,
heartbeatAt: (row.heartbeat_at as number | null) ?? 0,
leaseUntil: (row.lease_until as number | null) ?? 0,
}
}
@@ -289,3 +302,64 @@ export function toBusinessPlugin(row: Record<string, unknown>): BusinessPlugin {
updatedAt: row.updated_at as number,
}
}
// ── 集群化:worker 注册表与租约结果(v7;T08 S2 / 设计 §3.1–§3.2)──────────────
/** Worker 健康状态(`dsh_hosts.status` 的 CHECK 镜像)。 */
export type DshHostStatus = 'up' | 'draining' | 'down'
/** 一台承载用户实例的 worker(= 设计里的 Worker 节点)。 */
export interface DshHost {
id: string
/** agent 的内网地址,如 `10.0.1.11:9000`。 */
endpoint: string
/** 内部 HMAC 密钥(**只应存在于 DB 与 Manager 内存**,绝不经 API 返回)。 */
agentToken: string
/** 该机可用内存预算(MB);0 = 不承载实例(只做门户/控制)。 */
capacityMb: number
/** 由心跳上报的已用内存(MB)。 */
usedMb: number
status: DshHostStatus
/** 最近心跳(epoch 毫秒);null = 从未上报。 */
lastHeartbeat: number | null
}
/** Upsert payload for `dsh_hosts`(join 脚本/管理面用)。 */
export interface UpsertDshHostInput {
id: string
endpoint: string
agentToken: string
capacityMb: number
status?: DshHostStatus
}
/**
* 抢占结果。`ok:false` 时带回**当前持有者**与租约到期时刻,便于调用方决定
* "退让"还是"报告异常"(**不要据此接管** —— 见项目红线 R9)。
*/
export type ClaimResult =
| { ok: true; epoch: number; leaseUntil: number }
| { ok: false; holder: string | null; leaseUntil: number }
/**
* 集群模式下 main 实例行的**确定性 id**。
*
* 为什么需要确定性:租约是以 **(user, role='main')** 为单位的,`dsh_instances.id` 只是载体;
* 若每次 spawn 用随机 id,抢占时会插出多行 ⇒ 归属判断失效。local 模式仍用随机 id
* (它不写库),集群路径一律走这里。
*/
export function clusterInstanceId(userId: string): string {
return `dsh-${userId}`
}
export function toDshHost(row: Record<string, unknown>): DshHost {
return {
id: row.id as string,
endpoint: row.endpoint as string,
agentToken: row.agent_token as string,
capacityMb: (row.capacity_mb as number | null) ?? 0,
usedMb: (row.used_mb as number | null) ?? 0,
status: row.status as DshHostStatus,
lastHeartbeat: (row.last_heartbeat as number | null) ?? null,
}
}
+21 -3
View File
@@ -7,13 +7,31 @@
import type { ServerConfig } from '../config.js'
import { LocalUserFs } from './local-user-fs.js'
import { RemoteUserFs } from './remote-user-fs.js'
import type { UserFs } from './user-fs.js'
import { userRoot } from './workspace.js'
/** cluster 模式下的按用户路由(由 `server.ts` 注入;见 RemoteUserFsOptions 的说明)。 */
export interface ClusterFsRouting {
hostIdFor?: (userId: string) => Promise<string | undefined>
agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
}
/**
* Build the configured per-user filesystem. The single-machine backend
* touches the users volume in-process.
* Build the configured per-user filesystem.
* - `local` :控制面**在本进程内**直接碰用户卷(单机形态)。
* - `cluster` :用户卷在 **worker** 上 ⇒ 走 agent 的 `/fs/*`(T08 S5)——
* 远端实现复用同一份路径安全逻辑,`resolvePath` 按 **worker 的 dataRoot** 做路径数学。
*/
export function createUserFs(config: ServerConfig): UserFs {
export function createUserFs(config: ServerConfig, routing: ClusterFsRouting = {}): UserFs {
if (config.deployMode === 'cluster') {
return new RemoteUserFs({
agentUrl: config.clusterAgentUrl,
token: config.clusterAgentToken,
workerDataRoot: config.clusterWorkerDataRoot === '' ? config.dataRoot : config.clusterWorkerDataRoot,
hostIdFor: routing.hostIdFor,
agentFor: routing.agentFor,
})
}
return new LocalUserFs((userId) => userRoot(config.dataRoot, userId))
}
+180
View File
@@ -0,0 +1,180 @@
/**
* `UserFs` 的**远端实现**(T08 S5)。
*
* 为什么需要它:门户的「我的文件」(`/api/desktop/tree`、`/api/fs/*`) 只依赖 `UserFs` seam;
* 多机后用户卷在 **worker** 上,Manager 的本地读会落空。做法不是重新实现一套路径语义,
* 而是把 worker 上**同一个 `LocalUserFs`** 经 agent 的 `/fs/*` 暴露出来 —— 路径安全
* (`resolveWithinRoot` / `safeFilename` / `PathEscapeError`)**继续复用同一份代码**,
* 所以"本地能过的路径,远端行为一致"是结构性保证,不是靠测试碰运气。
*
* `resolvePath` 是**纯路径数学**(同步接口),按 **worker 的 dataRoot** 计算 ——
* 这正是"实例眼里的路径"。因此多机部署有一条**基线约定**:
* **所有 worker 的 dataRoot 必须是同一个绝对路径**(同镜像即可满足,见设计 §14.3 机器基线)。
* 不一致时 `buildServer` 会在启动时把差异**报出来**(见 `server.ts` 的 probe)。
*
* @module dshs/fs/remote-user-fs
*/
import { AGENT_TOKEN_HEADER } from '../worker/agent.js'
import { PathEscapeError, resolveWithinRoot } from '../web/middleware/fs-guard.js'
import type { PluginInfo } from './plugins.js'
import { UserFsError, isUserFsErrorCode, type UserFs } from './user-fs.js'
import type { FsEntry } from './workspace.js'
import { userRoot, workspaceRoot } from './workspace.js'
export interface RemoteUserFsOptions {
/** **默认/回退** worker agent 基址(未提供 hostIdFor 或查不到归属时用它)。 */
agentUrl: string
/** 与默认 agent 约定的共享密钥。 */
token: string
/**
* **按用户归属路由**(2026-09-15 生产切换暴露的缺口)。
*
* 为什么必须有:用户工作区在**那台 worker 的本地盘**上;文件面若固定打一台 agent,
* 就会出现「实例跑在 A、而 mkdir/上传写到 B」⇒ 实例看不到自己的文件、甚至 cwd 不存在而崩。
* 传 `hostIdFor`(查 `dsh_instances.host_id`)即可让每次文件操作落到**该用户所在的机器**。
*/
hostIdFor?: (userId: string) => Promise<string | undefined>
/** hostId → 接入信息(与 RemoteSpawner 用**同一份**目录,避免两套漂移)。 */
agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
/** **worker 上**的 dataRoot(必须与该 worker 一致,用于 `resolvePath` 的路径数学)。 */
workerDataRoot: string
/** 单次请求超时(ms)。文件可能较大,默认 30 s。 */
timeoutMs?: number
fetchImpl?: typeof fetch
}
export class RemoteUserFs implements UserFs {
private readonly base: string
private readonly token: string
/** **worker 上**的 dataRoot —— 公开只读,供 `server.ts` 启动时做基线一致性探测。 */
readonly workerDataRoot: string
private readonly timeoutMs: number
private readonly doFetch: typeof fetch
private readonly hostIdFor?: (userId: string) => Promise<string | undefined>
private readonly agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
constructor(options: RemoteUserFsOptions) {
this.base = options.agentUrl.replace(/\/$/, '')
this.token = options.token
this.hostIdFor = options.hostIdFor
this.agentFor = options.agentFor
this.workerDataRoot = options.workerDataRoot
this.timeoutMs = options.timeoutMs ?? 30_000
this.doFetch = options.fetchImpl ?? fetch
}
/** 解析该用户文件操作应打的那台 agent(查不到归属就用默认)。 */
private async target(userId: string): Promise<{ base: string; token: string }> {
if (this.hostIdFor === undefined) return { base: this.base, token: this.token }
try {
const hostId = await this.hostIdFor(userId)
if (hostId !== undefined && hostId !== null && this.agentFor !== undefined) {
const agent = this.agentFor(hostId)
if (agent !== undefined) return { base: agent.agentUrl.replace(/\/$/, ''), token: agent.token }
}
} catch {
// 查库失败 ⇒ 落回默认 host(宁可"可能读错机",也不要整个文件面 500)
}
return { base: this.base, token: this.token }
}
/** 统一的 POST:把 agent 的 `{error: code}` 还原成 `UserFsError`(路由按 code 回前端)。 */
private async post<T>(path: string, body: Record<string, unknown>): Promise<T> {
const userId = typeof body.userId === 'string' ? body.userId : ''
const t = await this.target(userId)
const res = await this.doFetch(`${t.base}${path}`, {
method: 'POST',
headers: { [AGENT_TOKEN_HEADER]: t.token, 'content-type': 'application/json' },
body: JSON.stringify(body),
signal: AbortSignal.timeout(this.timeoutMs),
})
const text = await res.text()
if (res.ok) return (text === '' ? undefined : JSON.parse(text)) as T
let code: unknown
try {
code = (JSON.parse(text) as { error?: unknown }).error
} catch {
code = undefined
}
if (typeof code === 'string' && isUserFsErrorCode(code)) throw new UserFsError(code)
throw new Error(`agent POST ${path} → ${res.status}: ${text.slice(0, 200)}`)
}
async initUserRoot(userId: string, uid?: number): Promise<void> {
// 注意:uid 也要带过去 —— 本地实现会 chown 用户根,远端由 worker 执行同一动作。
await this.post('/fs/init', uid === undefined ? { userId } : { userId, uid })
}
/**
* **纯路径数学**,按 worker 的 dataRoot 算(= 实例眼里的绝对路径)。
* 刻意**不**像本地实现那样 `ensureDir` —— Manager 不该在**自己**的盘上造目录。
*/
resolvePath(userId: string, relPath: string): string {
const ws = workspaceRoot(userRoot(this.workerDataRoot, userId))
try {
return resolveWithinRoot(ws, relPath)
} catch (err) {
if (err instanceof PathEscapeError) throw new UserFsError('bad_path')
throw err
}
}
async listDir(userId: string, relPath: string): Promise<FsEntry[]> {
return this.post<FsEntry[]>('/fs/list', { userId, relPath })
}
async mkdir(userId: string, relPath: string): Promise<void> {
await this.post('/fs/mkdir', { userId, relPath })
}
async createEntry(userId: string, relPath: string, name: string, type: 'file' | 'dir'): Promise<string> {
const out = await this.post<{ name: string }>('/fs/create', { userId, relPath, name, type })
return out.name
}
async upload(userId: string, relPath: string, name: string, data: Buffer): Promise<string> {
const out = await this.post<{ name: string }>('/fs/upload', {
userId,
relPath,
name,
dataBase64: data.toString('base64'),
})
return out.name
}
async isDirectory(userId: string, relPath: string): Promise<boolean> {
const out = await this.post<{ isDirectory: boolean }>('/fs/isdir', { userId, relPath })
return out.isDirectory
}
async readFile(userId: string, relPath: string, maxBytes?: number): Promise<{ name: string; data: Buffer }> {
const out = await this.post<{ name: string; dataBase64: string }>(
'/fs/read',
maxBytes === undefined ? { userId, relPath } : { userId, relPath, maxBytes },
)
return { name: out.name, data: Buffer.from(out.dataBase64, 'base64') }
}
async listInstalledPlugins(userId: string): Promise<PluginInfo[]> {
return this.post<PluginInfo[]>('/fs/plugins', { userId })
}
async writeHandoff(userId: string, content: string): Promise<void> {
await this.post('/fs/handoff', { userId, content })
}
/** 探测 worker 的 dataRoot(用于启动时的基线一致性检查,见 `server.ts`)。 */
async probeWorkerRoot(): Promise<string | undefined> {
try {
const res = await this.doFetch(`${this.base}/fs/root`, {
headers: { [AGENT_TOKEN_HEADER]: this.token },
signal: AbortSignal.timeout(5_000),
})
if (!res.ok) return undefined
const body = (await res.json()) as { dataRoot?: string }
return body.dataRoot
} catch {
return undefined
}
}
}
+169
View File
@@ -0,0 +1,169 @@
/**
* 实例归属租约(T08 S2;设计 §3.1–§3.2)。
*
* **为什么必须有它**:local 模式靠"进程内 Map + 单机"天然保证「一个用户同时只有一个活实例」;
* 多机后这个保证只能落到 **DB 的原子 CAS** 上 —— 否则两个 worker 会同时写同一个 `$DSH_HOME`
* (会话日志 append 冲突 ⇒ **数据损坏**,本库最贵的一类事故)。
*
* 三条不变量(都不许省):
* 1. **单写者**:抢占必须原子(`claimInstance` 的 `UPDATE … WHERE 无人持有 OR 租约过期`),
* 调用方以"是否真正更新到行"判定成败,**不许"先读后写"**。
* 2. **TTL > 2 × 续租间隔**:留足抖动余量;否则自身网络一抖就会误判自己失权
* (或更糟 —— 误判别人已死)。构造时**硬校验**,fail-loud。
* 3. **fencing**:每次操作带 `epoch`;不匹配 = 已被他人抢占 ⇒ 调用方必须**自杀**
* (self-fencing,如停掉自己那个实例),而不是继续写。
*
* ⚠️ 与 **R9** 的关系:本模块只提供"判定与递增 epoch"的机械能力,
* **不提供**"判定对方已死 → 接管"的自动化。没有心跳判据时单方面接管是被明令禁止的;
* 因此 `expiredAll()` 只用于**巡检/报告/人工确认后的动作**,绝不自动接管。
*
* @module dshs/supervisor/lease
*/
import type { DbAdapter } from '../db/adapter.js'
import type { ClaimResult, DshInstance } from '../db/types.js'
/** 默认租约存活 30 s(设计 §3.2 的建议时序)。 */
export const DEFAULT_LEASE_TTL_MS = 30_000
/** 默认续租间隔 10 s(不变量:ttl > 2 × renew)。 */
export const DEFAULT_LEASE_RENEW_MS = 10_000
export interface LeaseOptions {
/** 租约存活时长(ms)。不变量:必须 **> 2 × renewMs**。 */
ttlMs?: number
/** 续租间隔(ms)。 */
renewMs?: number
/** 注入时钟 —— 仅用于本类自己的判定(**SQL 里的 now 仍是 `Date.now()`**)。 */
now?: () => number
}
/**
* 某实例当前是否由「我」合法持有(fencing 判据)。
*
* 用在两处:① 收到心跳/指令前自检"我还是不是持有者"② 老 worker 复活后判断
* 自己**是否已被接管** ⇒ 是则自杀(防双写)。
*/
export function stillHolder(
instance: DshInstance | undefined,
hostId: string,
epoch: number,
): boolean {
return instance !== undefined && instance.hostId === hostId && instance.epoch === epoch
}
/** 一个用户实例的租约句柄(holding = 我持有 + 我的 epoch)。 */
export class InstanceLease {
readonly hostId: string
private readonly db: DbAdapter
private readonly ttl: number
private readonly renewInterval: number
private readonly now: () => number
/**
* userId → **我认领到的那把租约**(epoch + 落在哪个 host)。
*
* 为什么必须记住 hostId:多 worker 后 `renewInstanceLease(userId, hostId, epoch)` 要能在
* **正确的那个 host** 上校验;只记 epoch 会在多机下续错对象(2026-09-15 T08 S6)。
*/
private readonly held = new Map<string, { epoch: number; hostId: string }>()
constructor(db: DbAdapter, hostId: string, options: LeaseOptions = {}) {
this.db = db
this.hostId = hostId
this.ttl = options.ttlMs ?? DEFAULT_LEASE_TTL_MS
this.renewInterval = options.renewMs ?? DEFAULT_LEASE_RENEW_MS
this.now = options.now ?? Date.now
// 不变量:TTL 必须显著大于续租间隔,否则单次网络抖动就会造成"自己失权"或"误判他人已死"。
if (this.ttl <= 2 * this.renewInterval) {
throw new Error(
`lease: ttlMs(${this.ttl}) must be > 2 × renewMs(${this.renewInterval}) —— ` +
'否则时钟/网络抖动会破坏单写者保证(设计 §3.2 不变量 2)',
)
}
}
get ttlMs(): number {
return this.ttl
}
get renewMs(): number {
return this.renewInterval
}
/** 我当前认领的实例(userId → {epoch, hostId})。 */
holdings(): ReadonlyMap<string, { epoch: number; hostId: string }> {
return this.held
}
/**
* 抢占某用户 main 实例的归属。
*
* `ok:false` = 有人在管 ⇒ **退让**:不要接管、不要重试到死,交给上层决定
* (拉长等待 / 报告管理员)。成功时记住 epoch 供续租与 fencing 使用。
*/
async acquire(
userId: string,
hostId?: string,
meta?: { folder?: string; patch?: string },
): Promise<ClaimResult> {
const target = hostId ?? this.hostId
// meta(folder/patch)随认领一起落库 ⇒ 迁移才能复现启动参数
const res = await this.db.claimInstance(userId, target, this.ttl, meta)
if (res.ok) this.held.set(userId, { epoch: res.epoch, hostId: target })
return res
}
/**
* 续租。返回 false = **我已失权**(被他人以更高 epoch 抢占,或行被删)⇒ 调用方
* 必须 self-fence(停掉自己那个实例),并清掉本地记录。
*/
async renew(userId: string): Promise<boolean> {
const held = this.held.get(userId)
if (held === undefined) return false
const ok = await this.db.renewInstanceLease(userId, held.hostId, held.epoch, this.ttl)
if (!ok) this.held.delete(userId)
return ok
}
/** 批量续租(心跳 tick 用)。返回失权的 userId 列表(调用方据此 self-fence)。 */
async renewAll(): Promise<string[]> {
const lost: string[] = []
for (const userId of [...this.held.keys()]) {
if (!(await this.renew(userId))) lost.push(userId)
}
return lost
}
/** 主动释放(停实例时)。成功后不再持有该用户。 */
async release(userId: string): Promise<boolean> {
const held = this.held.get(userId)
if (held === undefined) return false
const ok = await this.db.releaseInstanceLease(userId, held.hostId, held.epoch)
if (ok) this.held.delete(userId)
return ok
}
/** 重新读取 DB 里的真实归属,校正本地记录(对账用)。 */
async refresh(userId: string): Promise<DshInstance | undefined> {
const inst = await this.db.findUserInstance(userId, 'main')
const held = this.held.get(userId)
if (!stillHolder(inst, held?.hostId ?? this.hostId, held?.epoch ?? -1)) this.held.delete(userId)
return inst
}
/** **我名下**租约已过期的实例(供巡检;**不等于可以接管**,见模块头与 R9)。 */
async expiredHere(): Promise<DshInstance[]> {
const all = await this.db.listExpiredInstanceLeases(this.now())
const mine = new Set([...this.held.values()].map((h) => h.hostId))
mine.add(this.hostId)
return all.filter((inst) => inst.hostId !== null && mine.has(inst.hostId))
}
/** 全集群租约已过期的实例(管理面/巡检用;**不自动接管**)。 */
async expiredAll(): Promise<DshInstance[]> {
return this.db.listExpiredInstanceLeases(this.now())
}
/** 对账:**一次拿回本机全部实例**(替代逐用户查询,设计 §11.6)。 */
async mine(): Promise<DshInstance[]> {
return this.db.listInstancesByHost(this.hostId)
}
}
+284
View File
@@ -0,0 +1,284 @@
/**
* 给任意 `Spawner` 套上**归属租约**(T08 S4;设计 §1.2/§3.2/§11.5)。
*
* 它补上集群模式下 Manager 侧最关键的一环:**"能不能拉起"必须先问过归属**。
* 本地模式靠"进程内 Map + 单机"天然保证单写者;多机后这个保证只能落在 DB 的原子 CAS 上
* —— 两个 Manager 各持一把租约、抢同一个用户,就是**双写同一个 home** = 数据损坏。
*
* 三个动作:
* 1. `launch` 前 **claim**:拿不到就抛 `LeaseBusyError`(**退让**,不是接管 —— 见 R9);
* 2. 心跳里 **renewAll**:续租;同时把"我已失权"的实例用 **`/fence`** 通知 worker 停掉
* (self-fencing 的 Manager 侧对齐,设计 §11.5);
* 3. `stop` 时 **release**:归属交还,别人立刻可以接管(不用等 TTL)。
*
* ⚠️ 归属只由 Manager 写(设计 §1.3 数据分层的判据 1)。worker 侧只被通知。
*
* @module dshs/supervisor/leased-spawner
*/
import { AGENT_TOKEN_HEADER } from '../worker/agent.js'
import type { DbAdapter } from '../db/adapter.js'
import type { Endpoint, Instance, Spawner, UserStatus } from './spawner.js'
import { InstanceLease, type LeaseOptions } from './lease.js'
/** 归属被别人持有时抛出 —— 调用方应**退让**(等待/报告),**不得接管**(R9)。 */
export class LeaseBusyError extends Error {
readonly userId: string
readonly holder: string | null
readonly leaseUntil: number
constructor(userId: string, holder: string | null, leaseUntil: number) {
super(`instance ${userId} is held by ${holder ?? 'someone'} until ${new Date(leaseUntil).toISOString()}`)
this.name = 'LeaseBusyError'
this.userId = userId
this.holder = holder
this.leaseUntil = leaseUntil
}
}
export interface LeasedSpawnerOptions extends LeaseOptions {
/** 本机(= 它所属的 worker)在 `dsh_hosts.id` 里的标识。 */
hostId: string
/** worker agent 基址。 */
agentUrl: string
/** 与 agent 约定的共享密钥。 */
agentToken: string
/** 心跳间隔(ms)。默认 = 续租间隔。 */
heartbeatMs?: number
/** 该 worker 的内存预算(MB);**0 = 不承载实例**(只做门户/控制,设计 §15.3)。 */
capacityMb?: number
/**
* **选机**(T08 S6):返回这次要把实例放到的 `hostId`。
*
* ⚠️ 实现必须**先粘性、再容量**(2026-09-15 生产切换时补的设计缺口):
* 用户的工作区是**跟机器走的**(本地盘)⇒ 把"已有历史数据在某台"的用户调度到另一台,
* 他打开实例会看到**空工作区**。所以:有历史归属且那台还 `up` ⇒ **留在原地**;
* 只有"从没有过归属"(新用户)才按容量挑最空的。
* 传入 `userId` 就是为了让实现能做这件事。
*/
selectHost?: (userId?: string) => Promise<string | undefined>
/** **hostId → agent 地址/密钥**(多机时 fence 要发给"实例所在的那台")。 */
agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
/**
* 启动时把自己注册进 `dsh_hosts`(幂等)。
*
* ⚠️ **一个 agent 只应有一条 host 记录**:`registerSelf` 只在"本 Manager 与 worker 同机"
* (1a 形态)时该开。**专用 Manager 部署必须关掉**(`DSHS_CLUSTER_REGISTER_SELF=0`),
* 否则会多出一条指向同一 agent 的 host 记录 ⇒ 同一个用户可能被两个 hostId 各自认领。
*/
registerSelf?: boolean
/** 关闭心跳(测试里手动 tick 用)。 */
manual?: boolean
}
/** 心跳里上报给 `dsh_hosts` 的本机观测值。 */
export interface HostObservation {
ok: boolean
instances: number
lastError?: string
}
export class LeasedSpawner implements Spawner {
private readonly lease: InstanceLease
private timer: NodeJS.Timeout | undefined
private lastObservation: HostObservation | undefined
private readonly heartbeatMs: number
constructor(
private readonly inner: Spawner,
private readonly db: DbAdapter,
private readonly options: LeasedSpawnerOptions,
) {
this.lease = new InstanceLease(db, options.hostId, options)
this.heartbeatMs = options.heartbeatMs ?? this.lease.renewMs
}
get hostId(): string {
return this.options.hostId
}
/** 最近一次心跳观测(管理面/诊断用)。 */
observation(): HostObservation | undefined {
return this.lastObservation
}
/** 本 Manager 当前持有的实例(userId → {epoch, hostId})。 */
holdings(): ReadonlyMap<string, { epoch: number; hostId: string }> {
return this.lease.holdings()
}
/** 注册本机 + 起心跳。窗口未开时先注册一次(否则管理面看不到这台 worker)。 */
async start(): Promise<void> {
if (this.options.registerSelf !== false) {
await this.db.upsertDshHost({
id: this.options.hostId,
endpoint: this.options.agentUrl,
agentToken: this.options.agentToken,
capacityMb: this.options.capacityMb ?? 0,
})
}
await this.tick() // 立即一次,管理面马上能看到心跳
if (this.options.manual === true) return
this.timer = setInterval(() => void this.tick(), this.heartbeatMs)
this.timer.unref?.()
}
/** 只停**心跳定时器**(不改实例)—— 注意别和 `Spawner.stop(userId)` 混淆,故另起名。 */
stopHeartbeat(): void {
if (this.timer !== undefined) {
clearInterval(this.timer)
this.timer = undefined
}
}
/** 一次心跳:续租 → 失权则 fence → 上报本机状态。 */
async tick(): Promise<void> {
// ① 续租;失权的 userId 会被清出本地记录
const lost = await this.lease.renewAll()
for (const userId of lost) {
// ② 我已失权 ⇒ 让 worker 停掉那个实例(下发的 epoch 取 DB 当前值 +1,确保高于它的记录)
await this.fenceOnAgent(userId)
}
await this.reportHost()
}
private async fenceOnAgent(userId: string): Promise<void> {
try {
const inst = await this.db.findUserInstance(userId, 'main')
// 多机(T08 S6):必须发给**实例所在的那台** —— 发错 host 等于没拦(旧持有者继续写)
const target = inst?.hostId === null || inst?.hostId === undefined
? { agentUrl: this.options.agentUrl, token: this.options.agentToken }
: (this.options.agentFor?.(inst.hostId) ?? { agentUrl: this.options.agentUrl, token: this.options.agentToken })
await this.post(`/fence`, { userId, epoch: (inst?.epoch ?? 0) + 1 }, target)
} catch {
// 通知失败不抛:下一轮心跳会重试;即便一直失败,租约已过期 ⇒ 新持有者会重建实例
}
}
private async reportHost(): Promise<void> {
const base = this.options.agentUrl.replace(/\/$/, '')
try {
const res = await fetch(`${base}/healthz`, { signal: AbortSignal.timeout(this.heartbeatMs) })
const body = (await res.json()) as { ok?: boolean; instances?: number }
this.lastObservation = { ok: body.ok === true, instances: body.instances ?? 0 }
await this.db.setDshHostStatus(
this.options.hostId,
this.lastObservation.ok ? 'up' : 'down',
undefined,
Date.now(),
)
} catch (err) {
this.lastObservation = { ok: false, instances: 0, lastError: err instanceof Error ? err.message : String(err) }
await this.db.setDshHostStatus(this.options.hostId, 'down', undefined, Date.now())
}
}
private async post(
path: string,
body: Record<string, unknown>,
target?: { agentUrl: string; token: string },
): Promise<unknown> {
const use = target ?? { agentUrl: this.options.agentUrl, token: this.options.agentToken }
const base = use.agentUrl.replace(/\/$/, '')
const res = await fetch(`${base}${path}`, {
method: 'POST',
headers: { [AGENT_TOKEN_HEADER]: use.token, 'content-type': 'application/json' },
body: JSON.stringify(body),
signal: AbortSignal.timeout(10_000),
})
if (!res.ok) throw new Error(`agent POST ${path} → ${res.status}`)
return res.json()
}
// ── Spawner 实现 ────────────────────────────────────────────────────────
async launch(
userId: string,
folder: string,
patch?: string,
opts?: { force?: boolean; epoch?: number; hostId?: string },
): Promise<Instance> {
// **先选机、再认领**(T08 S6):租约的 host_id 必须与"实例真正落在哪台"一致,
// 否则续租/释放会指向错误的对象(多机下就是静默的脑裂入口)。
// 显式 `opts.hostId`(迁移的目标机)优先于自动选机。
const hostId = opts?.hostId ?? (await this.options.selectHost?.(userId)) ?? this.options.hostId
const claim = await this.lease.acquire(userId, hostId, { folder, patch })
if (!claim.ok) throw new LeaseBusyError(userId, claim.holder, claim.leaseUntil)
try {
return await this.inner.launch(userId, folder, patch, { ...opts, epoch: claim.epoch, hostId })
} catch (err) {
// 拉起失败就**立刻交还归属** —— 否则要白等一个 TTL 才能重试(用户侧表现为"卡住")
await this.lease.release(userId)
throw err
}
}
async restartMain(userId: string): Promise<Instance | undefined> {
const current = await this.inner.status(userId)
if (current.main === undefined) return undefined
const { folder, patch } = current.main
await this.stop(userId)
return this.launch(userId, folder, patch)
}
/** 只重启**本 Manager 持有**的实例 —— 别人的归属不该被我重启(会与其持有者抢同一个 home)。 */
async restartAllMains(): Promise<void> {
for (const userId of [...this.lease.holdings().keys()]) {
try {
await this.restartMain(userId)
} catch {
// 单个失败不打断其余
}
}
}
async spawnWatchdog(userId: string): Promise<Instance | undefined> {
return this.inner.spawnWatchdog(userId)
}
async status(userId: string): Promise<UserStatus> {
return this.inner.status(userId)
}
async endpointFor(userId: string): Promise<Endpoint | undefined> {
return this.inner.endpointFor(userId)
}
async stop(userId: string, hostId?: string): Promise<void> {
// 显式 host 优先;否则用**我认领时那台**(认领记录里有)—— 别让 stop 落到别的 worker 上
const target = hostId ?? this.lease.holdings().get(userId)?.hostId
await this.inner.stop(userId, target)
await this.lease.release(userId)
}
async teardown(): Promise<void> {
this.stopHeartbeat()
await this.inner.teardown()
}
async waitForLaunchTokenForUser(userId: string, timeoutMs?: number): Promise<void> {
return this.inner.waitForLaunchTokenForUser(userId, timeoutMs)
}
async restartAndProbe(userId: string, settleMs?: number): Promise<{ ok: boolean; reason: string }> {
return this.inner.restartAndProbe(userId, settleMs)
}
touch(userId: string): void {
this.inner.touch(userId)
}
async ensureFileService(userId: string): Promise<void> {
return this.inner.ensureFileService(userId)
}
/**
* 透传可选观测面。`Spawner` 里这两个是**可选**方法 ⇒ 这里做条件委托:
* 内层有就转发(熔断/配额是 worker 本地自管的概念,设计 §1.2),没有就回 null。
*/
breakerInfo(userId: string): { opens: number; openedAt: number; cooldownUntil: number } | null {
return this.inner.breakerInfo?.(userId) ?? null
}
quotaInfo(userId: string): { baseMb: number; memMb: number; heapMb: number } | null {
return this.inner.quotaInfo?.(userId) ?? null
}
}
+103 -1
View File
@@ -22,7 +22,7 @@ import {
realpathSync,
writeFileSync,
} from 'node:fs'
import { join } from 'node:path'
import { dirname, join } from 'node:path'
import type { ServerConfig } from '../config.js'
import { handoffPath, homeRoot, userRoot, workspaceRoot } from '../fs/workspace.js'
import {
@@ -129,6 +129,47 @@ function withHeap(base: string | undefined, memMb: number): string {
}
/**
* 列出 `dest` 与 `stopAt` 之间的**祖先目录**(由外到内),用于 bwrap 的 `--tmpfs`(见
* {@link mountParentDirArgs}:`--tmpfs` 自带 0755,且**兼容 47 上的 bwrap 0.4.0**)。
*
* 背景(T08 S1.6,2026-09-15 实测):bwrap **只创建挂载点本身**,沿途缺失的父目录由它自建,
* 而权限是 **`0700 root:root`** —— 实测 `--bind /opt/a/b/c /opt/a/b/c` 会得到 `/opt`、
* `/opt/a`、`/opt/a/b` **全是 0700**。后果:**实例以非 root 的 uid 穿越这些路径时 EACCES**。
* 已实测到的两处症状:
* ① 宿主上 `/etc/ssl/openssl.cnf` 是指向 `/etc/pki/tls/openssl.cnf` 的**符号链接** ⇒ 解析要穿过
* `/etc/pki`(0700)⇒ node 报 `OpenSSL configuration error … Permission denied`、**exitCode 13**
* (106 / OpenCloudOS 9.6 实测;47 上**没有**该文件故静默跳过 ⇒ 同一份代码一台能跑一台崩);
* ② **用户工作区在沙箱内不可穿越** ⇒ 实例按**绝对路径**读写自己的文件被拒。
* 修法:把这些祖先目录**显式建成 0755**。**权限不扩大** —— 这些目录里只有随后绑定的白名单内容
* (整绑 `/etc/pki` 的替代方案已否决:会带入 `/etc/pki/tls/private/postfix.key`,违反 R5)。
*/
function mountParentDirList(dest: string, stopAt: string): string[] {
const dirs: string[] = []
let cur = dirname(dest)
while (cur !== stopAt && cur !== '/' && cur !== '' && cur !== '.') {
dirs.push(cur)
cur = dirname(cur)
}
return dirs.reverse() // 由外到内(`--perms` 只作用于紧接着的那一个 `--dir`)
}
/**
* 把 {@link mountParentDirList} 的结果摊平成 bwrap 参数。
* ⚠️ **不要再"就近调用"它**(例如插在 `--bind` 之前)—— 见 `bwrapArgs` 里"统一前置"
* 那段注释:就近创建会在嵌套前缀下遮掉已绑好的挂载点。当前实现只在**一处**统一使用。
*/
function mountParentDirArgs(dest: string, stopAt: string): string[] {
// ⚠️ **必须用 `--tmpfs`,不能用 `--perms 0755 --dir`**(2026-09-15 实测,差点打断生产):
// · `--perms` 是 bubblewrap **0.5+** 才有的选项;**47 上是 0.4.0** ⇒ 传了直接
// `bwrap: Unknown option --perms` ⇒ **沙箱起不来 = 所有实例全挂**(106 是 0.11.0,能过)。
// · 而 `--tmpfs` **自带 0755**(本文件另一处注释也这么写:`bwrap 的 --tmpfs 权限是 755`),
// 且在 0.4.0 上就可用。
// 47 上实测:改用 `--tmpfs` 后 `/etc` 可见条目 **78 → 78(零变化)**,且 `/etc/pki` 权限
// 由 `drwx------` 变为 `drwxr-xr-x`(可穿越)。代价 = 每个中间目录多一个空 tmpfs 挂载(极小)。
return mountParentDirList(dest, stopAt).flatMap((d) => ['--tmpfs', d])
}
/**
* Local backend: owns the lifecycle of per-user DSH process pairs via
* child_process. State is in-memory. Implements {@link Spawner}.
@@ -361,6 +402,26 @@ export class LocalSpawner implements Spawner {
return { main: this.mains.get(userId), watchdog: this.watchdogs.get(userId) }
}
/**
* 整机视角的实例清单(T08 S3:worker agent 的 `/instances` 用)。
*
* 口径 = **每个用户的 main 实例**(watchdog 是一次性 headless,不进对账口径)。
* 设计上这是 `/instances` "一次拿回整机"的实现,替代逐用户查询(设计 §11.6)。
*/
listUserInstances(): Instance[] {
return [...this.mains.values()]
}
/**
* 当前 main 实例的 launch token(T08 S3 P0-6)。
*
* 本地模式下 token 从子进程 stdout 解析出来;跨机后 **agent 必须把它回传 Manager**,
* 否则「登录直达会话」(档案 06/13/15)与实例侧 401 自愈(档案 24/50/51)都会失效。
*/
launchTokenOf(userId: string): string | undefined {
return this.mains.get(userId)?.launchToken
}
/** Endpoint the proxy forwards to (local → the running main's loopback port). */
async endpointFor(userId: string): Promise<Endpoint | undefined> {
const port = this.mains.get(userId)?.port
@@ -680,6 +741,30 @@ export class LocalSpawner implements Spawner {
'/etc/ssl',
]
const out: string[] = []
// 2026-09-15(T08 S1.6)**中间挂载点必须可穿越(0755)**。
//
// 现象(106 / OpenCloudOS 9.6 实测):实例起不来,子进程报
// `/usr/bin/node: OpenSSL configuration error: … Permission denied:
// … fopen(/etc/ssl/openssl.cnf, rb)` ⇒ **exitCode 13**。
// 根因:bwrap 会为 `--ro-bind-try /etc/pki/tls/certs …` 这类路径**自动补齐父目录**,
// 而这些自动创建的目录权限是 **0700(drwx------ root:root)** ⇒ 非 root 的实例
// **无法穿越**;宿主上 `/etc/ssl/openssl.cnf` 恰好是**指向 `/etc/pki/tls/openssl.cnf`
// 的符号链接** ⇒ 解析要穿过 `/etc/pki` → 被拒 → 报 **EACCES(不是 ENOENT)**
// → node 读 OpenSSL 配置**硬失败**。
// 为什么 47 没事:Alibaba Cloud Linux 3 上**没有** `/etc/ssl/openssl.cnf`
// ⇒ node 静默跳过 ⇒ **同一份代码一台能跑、一台崩**(机器基线差异,设计 §14.3)。
// 修法:在绑定**之前**把白名单路径在 `/etc` 下的所有中间目录显式建成 0755。
// 权限**不扩大**:`/etc` 在本沙箱里是 tmpfs,这些目录里只有下面白名单绑定的内容,
// 不新增任何宿主可见面。("整绑 `/etc/pki`"的替代方案已否决 —— 会顺带带入
// `/etc/pki/tls/private/postfix.key`,违反 **R5 权限只准收窄**。)
// 注意顺序:由外到内(内层挂载点要求外层已存在)。
const intermediates = new Set<string>()
for (const p of allow) {
for (const d of mountParentDirList(p, '/etc')) intermediates.add(d)
}
for (const d of [...intermediates].sort((a, b) => a.split('/').length - b.split('/').length)) {
out.push('--tmpfs', d) // 见 mountParentDirArgs 的注释:`--tmpfs` 自带 0755 且兼容 bwrap 0.4.0
}
for (const p of allow) {
let src = p
try {
@@ -691,6 +776,23 @@ export class LocalSpawner implements Spawner {
}
return out
})(),
// ── 所有挂载点的**中间目录**统一在这里建好(T08 S1.6 修正版)────────────────
// 为什么必须"统一前置 + 去重 + 由外到内"(2026-09-15 实测踩到的真 bug):
// 用户根(`--bind root root`)与共享技能层(`--ro-bind-try skill skill`)**可能嵌套在
// 同一前缀下**。若按"就近创建"把技能层的中间目录插在 `--bind root root` **之后**,
// 那么后挂的 `--tmpfs <共同祖先>` 会把**已经绑好的用户根整个遮掉** ⇒ bwrap 报
// `Can't chdir to <userRoot>/ws/xxx: No such file or directory` ⇒ 实例崩溃循环。
// 前置 + 去重后,中间目录只建一次,后续所有 bind 都落在它里面,谁也不遮谁。
// 权限不扩大:这些目录里只有随后绑定的白名单内容。
...(() => {
const dirs = new Set<string>()
for (const dest of [root, this.config.bundledSkillDir].filter((d) => d !== '')) {
for (const d of mountParentDirList(dest, '/')) dirs.add(d)
}
return [...dirs]
.sort((a, b) => a.split('/').length - b.split('/').length)
.flatMap((d) => ['--tmpfs', d])
})(),
'--dev', '/dev', '--proc', '/proc',
'--bind', tmpDir, '/tmp',
'--bind', root, root,
+330
View File
@@ -0,0 +1,330 @@
/**
* `Spawner` 的**远端实现**(T08 S3 单机 / S6 多机;设计 §1.1/§11)。
*
* 路由层只依赖 `Spawner` 接口(见 `spawner.ts` 头注释),所以本类**不触碰路由与代理**
* —— 代理层把 `endpointFor` 返回的 `{host, port}` 直连即可(`proxy.ts` 的 TCP 目标
* 与 Host 头本就是分开处理的,跨机不需要改信任逻辑)。
*
* 三条协议纪律:
* 1. **幂等键复用**:同一次调用的重试**复用同一个 `operationId`** —— 否则 agent 会把
* "Manager 超时后重发"当成新请求,起出两个实例(设计 §11.3 手段 3)。
* 2. **重试有界**:3 次(200ms / 1s / 3s),仍失败就**抛错**,由上层决定退让或告警;
* 错误信息里带上状态码与响应体片段,避免"静默失败"。
* 3. **按 host 路由**(S6):每个用户的操作都落到**它实例所在的那台** —— 依据是
* `dsh_instances.host_id`,由上层以 `hostIdFor` 注入(本类不直接连 DB)。
*
* ⚠️ 本类**不持有归属租约**:租约由 Manager 侧的 `InstanceLease`/`LeasedSpawner` 管理。
*
* @module dshs/supervisor/remote-spawner
*/
import { randomUUID } from 'node:crypto'
import { AGENT_TOKEN_HEADER } from '../worker/agent.js'
import type { Endpoint, Instance, Spawner, UserStatus } from './spawner.js'
/** 一台 worker 的接入信息。 */
export interface ClusterHost {
hostId: string
/** agent 基址。 */
agentUrl: string
/** 与 agent 约定的共享密钥。 */
token: string
/** 代理时使用的主机(同机 1a = `127.0.0.1`;跨机填 Worker 内网 IP)。 */
instanceHost?: string
}
export interface RemoteSpawnerOptions {
/** agent 基址,如 `http://127.0.0.1:9000`。 */
agentUrl: string
/** 与 agent 约定的共享密钥。 */
token: string
/** 代理时使用的主机(同机 1a = `127.0.0.1`;跨机填 Worker 内网 IP)。 */
instanceHost?: string
/** 单次请求超时(ms)。控制通道是短请求,默认 10 s(设计 §11.2)。 */
timeoutMs?: number
/** 注入 fetch(测试用)。 */
fetchImpl?: typeof fetch
/**
* 解析该用户的模型 key 与 uid —— Manager 侧解析后**随 launch 投递**给 agent。
* 为什么不投递"让 Worker 自己查凭据库":那是**权限扩大**(Worker 就能读全量用户的 key),
* 而投递是收窄到"本次实例那一把"。**注意**:这条只管**控制面凭据**,与"Worker 能不能有
* 自己的库"无关(插件数据在实例 home 里,见设计 §1.3 数据分层)。
*/
resolveApiKey?: (userId: string) => Promise<string | null>
resolveUid?: (userId: string) => Promise<number>
/** 默认 host 的 id(不提供 `hosts` 时的单机形态用它)。 */
defaultHostId?: string
/** **多机(S6)**:除默认 host 外的其它 worker。给了就按 `hostId` 路由。 */
hosts?: ClusterHost[]
/** **多机(S6)**:`userId` → 它实例所在的 `hostId`(上层查 `dsh_instances.host_id` 注入)。 */
hostIdFor?: (userId: string) => Promise<string | undefined>
/**
* **host 目录的来源**(S6):从 `dsh_hosts` 读。给它就**不必预知 worker 列表**,
* 且**新增 worker 无需重启 Manager**(TTL 内自动生效)。
*/
hostsProvider?: () => Promise<ClusterHost[]>
/** 目录缓存时长(ms)。默认 30 s —— 与心跳同量级。 */
directoryTtlMs?: number
}
const RETRY_DELAYS_MS = [200, 1000, 3000]
export class RemoteSpawner implements Spawner {
private readonly defaultHost: ClusterHost
private readonly hosts = new Map<string, ClusterHost>()
private readonly timeoutMs: number
private readonly doFetch: typeof fetch
private readonly resolveApiKey?: (userId: string) => Promise<string | null>
private readonly resolveUid?: (userId: string) => Promise<number>
private readonly hostIdFor?: (userId: string) => Promise<string | undefined>
private readonly hostsProvider?: () => Promise<ClusterHost[]>
private readonly directoryTtlMs: number
private directoryLoadedAt = 0
constructor(options: RemoteSpawnerOptions) {
this.defaultHost = {
hostId: options.defaultHostId ?? 'local',
agentUrl: options.agentUrl.replace(/\/$/, ''),
token: options.token,
instanceHost: options.instanceHost ?? '127.0.0.1',
}
this.hosts.set(this.defaultHost.hostId, this.defaultHost)
for (const host of options.hosts ?? []) {
this.hosts.set(host.hostId, { ...host, agentUrl: host.agentUrl.replace(/\/$/, '') })
}
this.timeoutMs = options.timeoutMs ?? 10_000
this.doFetch = options.fetchImpl ?? fetch
this.resolveApiKey = options.resolveApiKey
this.resolveUid = options.resolveUid
this.hostIdFor = options.hostIdFor
this.hostsProvider = options.hostsProvider
this.directoryTtlMs = options.directoryTtlMs ?? 30_000
}
/**
* 按需刷新 host 目录(TTL 内不重复查询)。
* **默认 host 始终在表里**(配置里那台),即使它还没注册进 `dsh_hosts`。
*/
private async ensureHosts(): Promise<void> {
if (this.hostsProvider === undefined) return
if (Date.now() - this.directoryLoadedAt < this.directoryTtlMs) return
this.directoryLoadedAt = Date.now()
try {
for (const host of await this.hostsProvider()) {
this.hosts.set(host.hostId, { ...host, agentUrl: host.agentUrl.replace(/\/$/, '') })
}
this.hosts.set(this.defaultHost.hostId, this.defaultHost)
} catch {
// 查库失败就沿用旧目录(可能是全库不可用的前兆,由心跳/管理面暴露)
}
}
/** 已知的 host 目录(管理面/诊断用)。 */
knownHosts(): ClusterHost[] {
return [...this.hosts.values()]
}
/** 强制刷新目录(管理面/测试用)。 */
async reloadHosts(): Promise<void> {
this.directoryLoadedAt = 0
await this.ensureHosts()
}
/** hostId → 接入信息;未知 host 回退到默认(单机形态下这就是唯一那台)。 */
hostById(hostId: string | null | undefined): ClusterHost {
if (hostId === null || hostId === undefined) return this.defaultHost
return this.hosts.get(hostId) ?? this.defaultHost
}
/**
* 决定这次操作落到哪台。优先级:**显式指定**(迁移目标机、launch 时选好的机)
* → **该用户实例的归属**(`hostIdFor`)→ 默认 host。
*/
private async hostFor(userId: string, explicit?: string): Promise<ClusterHost> {
await this.ensureHosts()
if (explicit !== undefined) {
let found = this.hosts.get(explicit)
if (found === undefined) {
// 目录有 30s TTL:显式指定的 host 可能"刚 join 还没进目录" ⇒ 强制刷一次
this.directoryLoadedAt = 0
await this.ensureHosts()
found = this.hosts.get(explicit)
}
// ⛔ 仍然找不到就**报错**,绝不回退到默认 host ——
// "租约认领在 A、实例却起在 B"是多机下最危险的静默失败(归属与实例分离)。
if (found === undefined) throw new Error(`unknown host "${explicit}":不在 host 目录里,拒绝改投到别的 worker`)
return found
}
if (this.hostIdFor !== undefined) {
const owned = await this.hostIdFor(userId)
if (owned !== undefined) {
const found = this.hosts.get(owned)
if (found !== undefined) return found
}
}
return this.defaultHost
}
/** 带重试的请求。`operationId` 由调用方生成并在重试间**保持不变**(幂等)。 */
private async call<T>(
host: ClusterHost,
method: 'GET' | 'POST',
path: string,
body?: Record<string, unknown>,
operationId?: string,
): Promise<T> {
const payload = body === undefined ? undefined : { ...body, ...(operationId === undefined ? {} : { operationId }) }
let lastErr: unknown
for (let attempt = 0; attempt <= RETRY_DELAYS_MS.length; attempt += 1) {
if (attempt > 0) await new Promise((r) => setTimeout(r, RETRY_DELAYS_MS[attempt - 1]))
try {
const res = await this.doFetch(`${host.agentUrl}${path}`, {
method,
headers: {
[AGENT_TOKEN_HEADER]: host.token,
...(payload === undefined ? {} : { 'content-type': 'application/json' }),
},
body: payload === undefined ? undefined : JSON.stringify(payload),
signal: AbortSignal.timeout(this.timeoutMs),
})
if (res.ok) return (await res.json()) as T
const text = await res.text()
// 4xx 是"协议/参数错",重试没意义;5xx 与网络错才重试。
if (res.status < 500) throw new Error(`agent ${method} ${path} → ${res.status}: ${text.slice(0, 200)}`)
lastErr = new Error(`agent ${method} ${path} → ${res.status}: ${text.slice(0, 200)}`)
} catch (err) {
lastErr = err
}
}
throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))
}
async launch(
userId: string,
folder: string,
patch?: string,
opts?: { force?: boolean; epoch?: number; hostId?: string },
): Promise<Instance> {
const host = await this.hostFor(userId, opts?.hostId)
// 同一个 operationId 贯穿这次调用的所有重试 ⇒ agent 侧幂等回放(不会起两个实例)。
const operationId = randomUUID()
const apiKey = this.resolveApiKey === undefined ? null : await this.resolveApiKey(userId)
const uid = this.resolveUid === undefined ? undefined : await this.resolveUid(userId)
const res = await this.call<{ instance: Instance; note?: string }>(
host,
'POST',
'/launch',
{ userId, folder, patch, apiKey, uid, epoch: opts?.epoch },
operationId,
)
return res.instance
}
async restartMain(userId: string, hostId?: string): Promise<Instance | undefined> {
const current = await this.status(userId)
if (current.main === undefined) return undefined
const { folder, patch } = current.main
const host = await this.hostFor(userId, hostId)
await this.stop(userId, host.hostId)
return this.launch(userId, folder, patch, { hostId: host.hostId })
}
/** 重启**所有 host 上**能看到的实例(跨机聚合;上层 `LeasedSpawner` 会限定在自己持有的范围内)。 */
async restartAllMains(): Promise<void> {
for (const host of this.hosts.values()) {
let instances: Array<{ userId: string }> = []
try {
const res = await this.call<{ instances: Array<{ userId: string }> }>(host, 'GET', '/instances')
instances = res.instances
} catch {
continue // 该 host 不可达:跳过(心跳/告警负责暴露)
}
for (const inst of instances) {
try {
await this.restartMain(inst.userId, host.hostId)
} catch {
// 单台失败不打断其余(与 LocalSpawner 的语义一致)
}
}
}
}
async spawnWatchdog(userId: string): Promise<Instance | undefined> {
const host = await this.hostFor(userId)
const res = await this.call<{ instance: Instance | null }>(host, 'POST', `/watchdog/${encodeURIComponent(userId)}`)
return res.instance ?? undefined
}
async status(userId: string): Promise<UserStatus> {
const host = await this.hostFor(userId)
const res = await this.call<{ main: Instance | null }>(host, 'GET', `/status/${encodeURIComponent(userId)}`)
return res.main === null ? {} : { main: res.main }
}
/** 代理目标:由**实例所在那台** agent 给端口(未运行 → undefined,代理会走冷启动分支)。 */
async endpointFor(userId: string): Promise<Endpoint | undefined> {
const host = await this.hostFor(userId)
const res = await this.call<{ running: boolean; host?: string; port?: number }>(
host,
'GET',
`/endpoint/${encodeURIComponent(userId)}`,
)
return res.running && res.host !== undefined && res.port !== undefined
? { host: res.host, port: res.port }
: undefined
}
async stop(userId: string, hostId?: string): Promise<void> {
const host = await this.hostFor(userId, hostId)
await this.call(host, 'POST', '/stop', { userId }, randomUUID())
}
async teardown(): Promise<void> {
// 远端实例的寿命长于任何单个 Manager 副本 ⇒ 由 Manager 的归属/租约管理,不在关闭时清。
}
/**
* 等 launch token 出现(本地模式 = 启动完成的信号)。
* 跨机后 token 由 agent 回传,语义不变;超时返回(不抛)—— 与本地实现一致。
*/
async waitForLaunchTokenForUser(userId: string, timeoutMs = 20_000): Promise<void> {
const deadline = Date.now() + timeoutMs
while (Date.now() < deadline) {
try {
const st = await this.status(userId)
if (st.main !== undefined && st.main.launchToken !== undefined) return
if (st.main === undefined) return // 没实例/已停 ⇒ 立即返回(与本地实现一致)
} catch {
// 网络抖动:继续等
}
await new Promise((r) => setTimeout(r, 200))
}
}
async restartAndProbe(userId: string, settleMs?: number): Promise<{ ok: boolean; reason: string }> {
const host = await this.hostFor(userId)
return this.call<{ ok: boolean; reason: string }>(
host,
'POST',
`/restart-probe/${encodeURIComponent(userId)}`,
settleMs === undefined ? {} : { settleMs },
)
}
/** 活动信号:转发给**实例所在那台** agent,让它自己的 idle-reap 不误杀(fire-and-forget)。 */
touch(userId: string): void {
void this.hostFor(userId)
.then((host) =>
this.doFetch(`${host.agentUrl}/touch/${encodeURIComponent(userId)}`, {
method: 'POST',
headers: { [AGENT_TOKEN_HEADER]: host.token },
}),
)
.catch(() => {
/* 活动信号丢了不影响正确性 */
})
}
async ensureFileService(_userId: string): Promise<void> {
// worker 本机就有用户卷(local 语义)⇒ 无需 sidecar。跨机文件面由 RemoteUserFs(/fs/*)承担。
}
}
+17 -2
View File
@@ -92,7 +92,22 @@ export interface Endpoint {
* `LocalSpawner` materializes it to a file inside the user's own volume.
*/
export interface Spawner {
launch(userId: string, folder: string, patch?: string, opts?: { force?: boolean }): Promise<Instance>
/**
* 拉起实例。
*
* `opts.epoch`(T08 S4):**集群模式下 epoch 是 launch 契约的一部分** —— Manager 先抢占
* 归属拿到 epoch,再把它随 launch 下发,worker 记下来用于 **self-fencing**
* (收到更高 epoch 就停掉自己那个实例)。本地模式忽略该字段。
*
* `opts.hostId`(T08 S6):**多 worker 时指定落到哪台** —— 由上层选好机、并已用它认领租约,
* 因此这里必须与租约的 `host_id` 一致(否则归属与实例分离)。单机/1a 忽略。
*/
launch(
userId: string,
folder: string,
patch?: string,
opts?: { force?: boolean; epoch?: number; hostId?: string },
): Promise<Instance>
restartMain(userId: string): Promise<Instance | undefined>
/**
* 档案 78:熔断观测面(可选 —— 熔断是**本地模式**概念,k8s 模式没有)。
@@ -112,7 +127,7 @@ export interface Spawner {
spawnWatchdog(userId: string): Promise<Instance | undefined>
status(userId: string): Promise<UserStatus>
endpointFor(userId: string): Promise<Endpoint | undefined>
stop(userId: string): Promise<void>
stop(userId: string, hostId?: string): Promise<void>
teardown(): Promise<void>
/** 等待该用户 main 实例打印 launch token(本地模式 = 启动完成的信号)。无实例 /
* 已崩溃 / 已停 → 立即返回;k8s 模式无 token 概念 → no-op。供 enter 复用分支在返回
+97
View File
@@ -155,4 +155,101 @@ export const adminRoutes: FastifyPluginAsync = async (app) => {
}
})
// ── 集群管理面(T08 S6):worker 注册表 + 实例迁移 ────────────────────────
/**
* Worker 列表。
* ⚠️ **绝不下发 `agentToken`** —— 它是内网共享密钥,只回 `hasToken` 供排查"配没配"。
*/
app.get('/api/admin/hosts', { preHandler: requireAdmin }, async () => {
const hosts = await app.db.listDshHosts()
return {
deployMode: app.config.deployMode,
hosts: hosts.map((h) => ({
id: h.id,
endpoint: h.endpoint,
capacityMb: h.capacityMb,
usedMb: h.usedMb,
status: h.status,
lastHeartbeat: h.lastHeartbeat,
hasToken: h.agentToken !== '',
})),
}
})
/** 注册/更新一台 worker(join 脚本调用;**幂等**:同 id 重复执行 = 更新并标回 `up`)。 */
app.post('/api/admin/hosts', { preHandler: requireAdmin }, async (request, reply) => {
const body = request.body as {
id?: string
endpoint?: string
token?: string
capacityMb?: number
}
if (body.id === undefined || body.endpoint === undefined || body.token === undefined) {
return reply.code(400).send({ error: 'id, endpoint and token are required' })
}
const host = await app.db.upsertDshHost({
id: body.id,
endpoint: body.endpoint,
agentToken: body.token,
capacityMb: Number(body.capacityMb ?? 0),
})
await app.db.audit(request.user?.id ?? null, 'host.upsert', JSON.stringify({ id: host.id, endpoint: host.endpoint }))
return {
ok: true,
host: { id: host.id, endpoint: host.endpoint, capacityMb: host.capacityMb, status: host.status },
}
})
/**
* **计划内迁移**(T08 S6;设计 §4.2):drain → 目标机拉起 → 归属原子更新(epoch+1)。
*
* 顺序不可换:**先停源、再在目标机拉起**。如果反序,两台上会同时有实例(同一个 home ⇒ 双写)。
* 归属的原子性由租约保证(`claimInstance` 会把 `host_id` 换成目标机并 `epoch+1`),
* 所以旧机即便复活也会被 fencing 挡住(设计 §11.5)。
*
* ⚠️ **数据不搬家**:`folder` 是实例眼里的绝对路径,能在目标机上生效的前提是
* **两台 worker 的 dataRoot 同路径 + 用户数据位置无关**(共享存储或已同步)——
* 这正是设计 §12/§14.3 的前提,不是本路由能替你保证的。
*/
app.post('/api/admin/users/:id/dsh/migrate', { preHandler: requireAdmin }, async (request, reply) => {
const { id } = request.params as { id: string }
const { targetHost } = request.body as { targetHost?: string }
if (targetHost === undefined || targetHost === '') {
return reply.code(400).send({ error: 'targetHost is required' })
}
const before = await app.db.findUserInstance(id, 'main')
if (before === undefined) return reply.code(404).send({ error: 'not_found', detail: '该用户没有 main 实例记录' })
if (before.hostId === targetHost) return reply.code(409).send({ error: 'already_there' })
const target = await app.db.findDshHost(targetHost)
if (target === undefined) return reply.code(404).send({ error: 'unknown_host' })
if (target.status === 'down') return reply.code(409).send({ error: 'target_down' })
const source = before.hostId
// fail-loud:没有 folder 就没法在目标机上复现启动(空 cwd 会让 bwrap 直接崩)
if ((before.folder ?? '') === '') {
return reply.code(409).send({ error: 'no_folder_recorded', detail: '该实例没有记录 folder,无法复现启动' })
}
// ① drain:停源机实例(优雅停机 → 会话落盘;同时释放归属)
if (source !== null) await app.supervisor.stop(id, source)
// ② 目标机拉起(走租约:以 targetHost 认领 → epoch+1)
const instance = await app.supervisor.launch(id, before.folder ?? '', before.patch ?? undefined, {
hostId: targetHost,
})
const after = await app.db.findUserInstance(id, 'main')
await app.db.audit(
request.user?.id ?? null,
'dsh.migrate',
JSON.stringify({ userId: id, from: source, to: after?.hostId, epoch: after?.epoch }),
)
return {
ok: true,
from: source,
to: after?.hostId ?? null,
epoch: after?.epoch ?? 0,
port: instance.port ?? null,
}
})
}
+144 -2
View File
@@ -13,10 +13,13 @@ import { chown, mkdir, readFile, stat, writeFile } from 'node:fs/promises'
import type { ServerConfig } from '../config.js'
import { createDbAdapter, type CredentialLandingRow, type DbAdapter, type PublicUser } from '../db/index.js'
import { createUserFs } from '../fs/provider.js'
import { RemoteUserFs } from '../fs/remote-user-fs.js'
import type { UserFs } from '../fs/user-fs.js'
import { decrypt, deriveKey } from '../crypto.js'
import { hashUid } from '../isolation.js'
import { LocalSpawner } from '../supervisor/orchestrator.js'
import { LeasedSpawner } from '../supervisor/leased-spawner.js'
import { RemoteSpawner, type ClusterHost } from '../supervisor/remote-spawner.js'
import { registerDshProxy } from '../supervisor/proxy.js'
import type { Spawner } from '../supervisor/spawner.js'
import {
@@ -259,8 +262,147 @@ export async function buildServer(config: ServerConfig): Promise<FastifyInstance
if (config.deployMode === 'k8s') {
throw new Error('deployMode "k8s" is not supported by this build: only the single-machine backend ships')
}
const supervisor: Spawner = new LocalSpawner(config, resolveApiKey, resolveUid)
const userFs = createUserFs(config)
// cluster 模式(T08 S3/S4):实例在 worker 上,Manager 只投递操作 + 代理。
// fail-loud:没配 agent 地址就直接报错,别等第一个用户点进来才发现。
if (config.deployMode === 'cluster' && config.clusterAgentUrl === '') {
throw new Error('deployMode=cluster requires DSHS_CLUSTER_AGENT_URL (e.g. http://127.0.0.1:9000)')
}
// cluster:RemoteSpawner(传输)+ LeasedSpawner(**归属租约**)——
// 后者保证"能不能拉起先问归属",这是多机下防双写同一个 home 的承重件(设计 §3.2)。
let leased: LeasedSpawner | undefined
// ── 多 worker 的 host 目录(T08 S6)────────────────────────────────────
// 由 `dsh_hosts` 派生并**随用随刷新**(TTL 30 s)⇒ **新增 worker 不必重启 Manager**。
// 同时供三处使用:RemoteSpawner 的按 host 路由、LeasedSpawner 的 fence 目标、
// 以及 `selectHost` 的容量准入 —— 都读**同一份**内存目录,避免三套各自漂移。
const hostDirectory = new Map<string, ClusterHost>()
hostDirectory.set(config.clusterHostId, {
hostId: config.clusterHostId,
agentUrl: config.clusterAgentUrl,
token: config.clusterAgentToken,
instanceHost: config.clusterInstanceHost,
})
const hostsProvider = async (): Promise<ClusterHost[]> => {
for (const row of await db.listDshHosts()) {
hostDirectory.set(row.id, {
hostId: row.id,
agentUrl: row.endpoint,
token: row.agentToken,
instanceHost: config.clusterInstanceHost,
})
}
return [...hostDirectory.values()]
}
/**
* 按用户归属解析 host(实例面与**文件面**共用这一份,避免两套路由漂移)。
*
* 为什么两处都要用:用户工作区在**那台 worker 的本地盘**;若文件面固定打一台 agent,
* 就会出现「实例跑在 A、mkdir/上传写到 B」⇒ 实例看不到自己的文件、甚至 cwd 不存在而崩
* (2026-09-15 生产切换暴露)。
*/
const hostIdForUser = async (userId: string): Promise<string | undefined> =>
(await db.findUserInstance(userId, 'main'))?.hostId ?? undefined
/**
* **文件面专用**路由:没有归属就**先选机并钉住**。
*
* 为什么不能直接用 hostIdForUser:新用户还没有归属,"写文件"和"launch"会各自选一次机,
* 两次可能选到不同机器 ⇒「文件写到 A、实例起在 B」⇒ 实例看不到自己的文件(2026-09-15 实测)。
* 首次触达工作区就把归属钉住,后续(含 launch)全走粘性 ⇒ 两面必然一致。
*/
const hostIdForFile = async (userId: string): Promise<string | undefined> => {
const owned = await hostIdForUser(userId)
if (owned !== undefined && owned !== null) return owned
const chosen = (await selectHost(userId)) ?? config.clusterHostId
if (chosen === '') return undefined
await db.pinInstanceHost(userId, chosen)
return chosen
}
/**
* 选机:**① 粘性优先 ② 再按容量准入**。
*
* ⚠️ 顺序不能颠倒(2026-09-15 生产切换时补的缺口):用户工作区在**本地盘**、跟着机器走,
* 把"已有历史数据的用户"调度到另一台 ⇒ 他打开实例看到**空工作区**。
* ⇒ 有历史归属且那台还 `up` 就留在原地;只有**从未有过归属**(新用户)才按容量挑最空的。
* `capacityMb <= 0` = 未声明(不设限);`-1` = 显式禁用承载。
*/
const reserveMb = Number(process.env.DSHS_CLUSTER_RESERVE_MB ?? '512')
const selectHost = async (userId?: string): Promise<string | undefined> => {
const rows = await db.listDshHosts()
const eligible = rows.filter((h) => h.status === 'up' && h.capacityMb !== -1)
if (userId !== undefined) {
const owned = (await db.findUserInstance(userId, 'main'))?.hostId ?? null
if (owned !== null && eligible.some((h) => h.id === owned)) return owned
}
const candidates = eligible.filter(
(h) => h.capacityMb <= 0 || h.usedMb + reserveMb <= h.capacityMb,
)
if (candidates.length === 0) return undefined // 无候选 ⇒ 回退到配置里那台
candidates.sort((a, b) => a.usedMb - b.usedMb)
return candidates[0].id
}
const supervisor: Spawner =
config.deployMode === 'cluster'
? (leased = new LeasedSpawner(
new RemoteSpawner({
agentUrl: config.clusterAgentUrl,
token: config.clusterAgentToken,
instanceHost: config.clusterInstanceHost,
defaultHostId: config.clusterHostId,
hostsProvider,
resolveApiKey,
resolveUid,
// 按 host 路由:每次操作都落到"该用户实例所在那台"(与文件面同一份)
hostIdFor: hostIdForUser,
}),
db,
{
hostId: config.clusterHostId,
agentUrl: config.clusterAgentUrl,
agentToken: config.clusterAgentToken,
capacityMb: Number(process.env.DSHS_CLUSTER_CAPACITY_MB ?? '0'),
// 专用 Manager 部署设 DSHS_CLUSTER_REGISTER_SELF=0(见 LeasedSpawner 的注释)
registerSelf: (process.env.DSHS_CLUSTER_REGISTER_SELF ?? '1') !== '0',
ttlMs: Number(process.env.DSHS_CLUSTER_LEASE_TTL_MS ?? '30000'),
renewMs: Number(process.env.DSHS_CLUSTER_LEASE_RENEW_MS ?? '10000'),
selectHost,
agentFor: (hostId: string) => {
const h = hostDirectory.get(hostId)
return h === undefined ? undefined : { agentUrl: h.agentUrl, token: h.token }
},
},
))
: new LocalSpawner(config, resolveApiKey, resolveUid)
// 注册本机 + 起心跳(异步,不阻塞启动;心跳失败只影响该 worker 的状态位)
if (leased !== undefined) {
void leased.start().catch((err: unknown) => {
console.error('[cluster] heartbeat/register failed to start:', err)
})
}
const userFs = createUserFs(config, {
hostIdFor: hostIdForFile,
agentFor: (hostId: string) => {
const h = hostDirectory.get(hostId)
return h === undefined ? undefined : { agentUrl: h.agentUrl, token: h.token }
},
})
// T08 S5:cluster 模式下**所有 worker 的 dataRoot 必须是同一绝对路径**(基线约定,
// 设计 §14.3)。不一致会让 `resolvePath` 算出的"实例眼里的路径"与实际不符 ⇒
// 文件面与 launch 的 folder 都会错。这里在启动时**报出来**,别等用户点进去才发现。
if (userFs instanceof RemoteUserFs) {
void userFs
.probeWorkerRoot()
.then((root) => {
if (root !== undefined && root !== userFs.workerDataRoot) {
console.error(
`[cluster] worker dataRoot 与配置不一致:agent 报 ${root},本进程按 ${userFs.workerDataRoot} 计算路径。` +
'请把 DSHS_CLUSTER_WORKER_DATA_ROOT 设为 worker 上的实际值(所有 worker 必须同路径)。',
)
}
})
.catch(() => {
/* 探测失败不阻塞启动:会有心跳/调用失败暴露 */
})
}
const app = Fastify({
logger: { level: config.logLevel },
+450
View File
@@ -0,0 +1,450 @@
/**
* Worker agent(T08 S3;设计 §11.2)。
*
* **它是什么**:Worker 上唯一的"被拨入口" —— 一个内部 HTTP 服务,把实例生命周期
* 暴露给 Manager。**它不做归属决策**(谁托管谁是 Manager + PG 的事),只负责
* "在这台机器上把实例起停好",并复用 **`LocalSpawner`**,因此 bwrap/uid/scope
* 隔离、内存配额推导、崩溃退避与熔断、插件探活这些**本地语义全部原样保留**
* (这是本方案相对 k8s 路线最大的成本优势)。
*
* 四条协议纪律(设计 §11.3):
* 1. **单向拨入**:Worker 不反向连 Manager、不写控制面数据(自己可以有库,见下);
* 2. **幂等键**:每个变更请求带 `operationId`,重复请求**回放上次结果**
* (否则 Manager 超时重试会起两个实例);
* 3. **最小接口**:只接受白名单动作,参数受限(folder 由 Manager 解析、patch 有长度上限)
* —— agent 若能被当任意命令执行器,Worker 沦陷 = 全集群沦陷;
* 4. **self-fencing**:`POST /fence {userId, epoch}` —— 本地记录的 epoch 落后于
* Manager 下发的值 ⇒ **主动停掉该实例**(防双写的最后一道防线)。
*
* @module dshs/worker/agent
*/
import { timingSafeEqual } from 'node:crypto'
import Fastify, { type FastifyInstance, type FastifyReply } from 'fastify'
import type { ServerConfig } from '../config.js'
import { LocalUserFs } from '../fs/local-user-fs.js'
import { userRoot } from '../fs/workspace.js'
import { isUserFsErrorCode, UserFsError } from '../fs/user-fs.js'
import { hashUid } from '../isolation.js'
import { LocalSpawner } from '../supervisor/orchestrator.js'
import { SshTunnel } from './tunnel.js'
import type { Instance } from '../supervisor/spawner.js'
/** 绑定的头部名(Manager/agent 双方约定)。 */
export const AGENT_TOKEN_HEADER = 'x-dsh-agent-token'
export interface WorkerAgentOptions {
/** 本机在 `dsh_hosts.id` 里的标识。 */
hostId: string
/** 共享密钥(仅内网 + nft 白名单;本版是 bearer 式比较,HMAC/防重放留待后续)。 */
token: string
/** 监听端口。 */
port: number
/** 绑定地址(默认 `0.0.0.0`,靠 nft 只放行 Manager 网段)。 */
host?: string
/** 返回给 Manager 做代理的地址(同机 1a 用 `127.0.0.1`;跨机时填内网 IP)。 */
instanceHost?: string
/** 日志级别。 */
logLevel?: string
/**
* **反向隧道**(跨机演练):Worker 主动拨 Manager,形如 `[email protected]:32022`。
* 不设则完全关闭(同机/单机形态零影响)。见 `tunnel.ts` 头注释。
*/
tunnelTarget?: string
/** 隧道私钥(默认 `~/.ssh/tunnel_ed25519`)。 */
tunnelIdentity?: string
/** ControlMaster socket(默认 `/tmp/dshs-tunnel-<hostId>.sock`)。 */
tunnelControlPath?: string
}
/** 变更类请求的幂等缓存条数上限(超出后丢最旧的 —— 只是省重试,不是审计)。 */
const OP_CACHE_MAX = 512
/** patch 内容长度上限(防把 agent 当大对象存储)。 */
const MAX_PATCH_BYTES = 256 * 1024
interface OpCache {
order: string[]
results: Map<string, unknown>
}
/**
* 组装 agent。**复用 `LocalSpawner`**(隔离/配额/退避/熔断/探活全部原样保留)。
*
* ⚠️ **边界要读准**(2026-09-15 用户纠正):本 agent **不写控制面数据**(尤其归属/租约 ——
* 双写就是脑裂),所以 `apiKey` 与 `uid` **不由本机查控制面库**,而是 Manager 在
* `POST /launch` 时随请求投递(与 k8s 用 per-user Secret 同一思路),只存内存。
* 但这**不等于"Worker 不许有数据库"**:插件的 per-user 数据(如 `home/.dsh/mcn-plugin.db`)
* 属于**实例业务数据**,由实例自己读写、跟着 home 走;Worker 也可以有自己的运维库。
* 完整判据见设计 §1.3「数据分层」。
*/
export function buildWorkerAgent(
config: ServerConfig,
options: WorkerAgentOptions,
): { app: FastifyInstance; spawner: LocalSpawner; stop: () => Promise<void> } {
/** launch 时投递、仅存内存的凭据与 uid(Worker 不连 DB)。 */
const apiKeys = new Map<string, string>()
const uids = new Map<string, number>()
const spawner = new LocalSpawner(
config,
async (userId: string) => apiKeys.get(userId) ?? null,
async (userId: string) => uids.get(userId) ?? hashUid(userId, config.baseUid),
)
/**
* 文件面(T08 S5):**复用同一个 `LocalUserFs`** —— 用户卷本来就在本机,
* 所以"跨机文件面"= 把这个实现经 HTTP 暴露出去,而不是重新实现一套路径语义。
*/
const userFs = new LocalUserFs((userId: string) => userRoot(config.dataRoot, userId))
/**
* 反向隧道(可选)。静态转发 = **agent 自身端口** + `DSHS_TUNNEL_STATIC_PORTS`(如控制面 PG);
* 实例端口在 launch/stop 时动态加减,并在 `/healthz`(Manager 的心跳)里**对账自愈**。
*/
const tunnelTarget = options.tunnelTarget ?? process.env.DSHS_TUNNEL_TARGET ?? ''
const staticPorts = [
options.port,
...(process.env.DSHS_TUNNEL_STATIC_PORTS ?? '')
.split(',')
.map((v) => Number(v.trim()))
.filter((v) => Number.isInteger(v) && v > 0),
]
const tunnel =
tunnelTarget === ''
? undefined
: new SshTunnel({
target: tunnelTarget,
identity:
options.tunnelIdentity ??
process.env.DSHS_TUNNEL_IDENTITY ??
`${process.env.HOME ?? '/root'}/.ssh/tunnel_ed25519`,
controlPath: options.tunnelControlPath ?? `/tmp/dshs-tunnel-${options.hostId}.sock`,
staticPorts,
})
let tunnelReady = tunnel === undefined
/**
* **隧道自愈**:master 失联就重建(重建会自动补回 staticPorts 的静态转发)。
*
* ⚠️ 为什么不能只在 `/healthz` 里做(2026-09-15 想清楚的一个死角):`/healthz` 是**经隧道**
* 才打得进来的 —— 隧道一断,Manager 的心跳就进不来,自愈**永远不会被触发**(自己把自己锁死)。
* ⇒ 必须由 **agent 本地定时器**驱动(下面 20s 一跳),`/healthz` 里再顺手做一次。
*/
const healTunnel = async (): Promise<void> => {
if (tunnel === undefined) return
if (tunnelReady && (await tunnel.isMasterAlive())) return
tunnelReady = false
try {
await tunnel.ensureMaster()
tunnelReady = true
console.error('[tunnel] master 失联 → 已重建(含静态转发)')
} catch (err) {
console.error('[tunnel] 重建失败,下轮再试:', err instanceof Error ? err.message : err)
}
}
/** 把活着的实例端口补齐、把已消失的撤掉(崩溃退出也走这里收敛,不必逐个挂 exit 钩子)。 */
const reconcileTunnel = async (): Promise<void> => {
if (tunnel === undefined || !tunnelReady) return
const live = new Set((await spawner.listUserInstances()).map((i) => i.port).filter((p): p is number => p !== undefined))
for (const port of live) await tunnel.forward(port)
for (const port of tunnel.ports) {
if (!live.has(port) && !staticPorts.includes(port)) await tunnel.cancel(port)
}
}
let tunnelTimer: NodeJS.Timeout | undefined
if (tunnel !== undefined) {
void tunnel
.ensureMaster()
.then(() => {
tunnelReady = true
})
.catch((err: unknown) => {
console.error('[tunnel] 建立失败(跨机代理将不可用,本机功能不受影响):', err instanceof Error ? err.message : err)
})
tunnelTimer = setInterval(() => {
void healTunnel().then(reconcileTunnel)
}, 20_000)
tunnelTimer.unref?.()
}
const app = Fastify({ logger: { level: options.logLevel ?? 'info' }, bodyLimit: MAX_PATCH_BYTES + 4096 })
const cache: OpCache = { order: [], results: new Map() }
/** agent 侧记住的 epoch(self-fencing 判据)。 */
const epochs = new Map<string, number>()
const remember = (op: string, value: unknown): void => {
if (cache.results.has(op)) return
cache.results.set(op, value)
cache.order.push(op)
while (cache.order.length > OP_CACHE_MAX) {
const oldest = cache.order.shift()
if (oldest !== undefined) cache.results.delete(oldest)
}
}
app.addHook('onRequest', async (request, reply) => {
if (request.url === '/healthz') return // 存活探测不带凭据
const given = request.headers[AGENT_TOKEN_HEADER]
const expected = options.token
const a = Buffer.from(typeof given === 'string' ? given : '')
const b = Buffer.from(expected)
if (a.length !== b.length || !timingSafeEqual(a, b)) {
await reply.code(401).send({ error: 'unauthorized' })
}
})
app.get('/healthz', async () => {
const instances = await spawner.listUserInstances()
// 心跳里顺手自愈 + 对账(主驱动是本地定时器,见 healTunnel 的注释)
await healTunnel()
await reconcileTunnel()
return {
ok: true,
hostId: options.hostId,
instances: instances.length,
tunnel: tunnel === undefined ? null : { ready: tunnelReady, ports: tunnel.ports },
}
})
/** 对账用:**一次拿回整机**(设计 §11.6,替代逐用户查询)。 */
app.get('/instances', async () => ({ instances: await spawner.listUserInstances() }))
app.post('/launch', async (request, reply) => {
const body = request.body as {
userId?: string
folder?: string
patch?: string
epoch?: number
operationId?: string
apiKey?: string | null
uid?: number
}
if (body.userId === undefined || body.operationId === undefined) {
return reply.code(400).send({ error: 'userId and operationId are required' })
}
const cached = cache.results.get(body.operationId)
if (cached !== undefined) return cached // 幂等回放
// 先落凭据/uid/**epoch**(重放路径也安全:同值覆盖)。
// ⚠️ epoch 记录的是「Manager 的意图」,因此必须在**尝试 spawn 之前**落 ——
// "实例本来就在跑"(AlreadyRunningError 分支)时也要记,否则 `/fence` 拿不到
// 我的 epoch,self-fencing 就永远不触发(2026-09-15 T08 S3 实测踩到)。
if (body.apiKey !== undefined && body.apiKey !== null) apiKeys.set(body.userId, body.apiKey)
if (body.uid !== undefined) uids.set(body.userId, body.uid)
if (body.epoch !== undefined) epochs.set(body.userId, body.epoch)
try {
const instance = await spawner.launch(body.userId, body.folder ?? '', body.patch)
// 跨机:把该实例端口经隧道打到 Manager 侧(失败不阻断 —— 本机仍可用)
if (tunnel !== undefined && tunnelReady && instance.port !== undefined) await tunnel.forward(instance.port)
const payload = { instance: { ...instance, launchToken: spawner.launchTokenOf(body.userId) } }
remember(body.operationId, payload)
return payload
} catch (err) {
// 已在跑:**返回现有实例**而不是报错 —— 这让重试天然安全(与 AlreadyRunningError 语义对齐)。
const msg = err instanceof Error ? err.message : String(err)
if (/already has a running/i.test(msg)) {
const currents = await spawner.listUserInstances()
const found = currents.find((i) => i.userId === body.userId)
if (found !== undefined) {
const payload = { instance: { ...found, launchToken: spawner.launchTokenOf(body.userId) }, note: 'already-running' }
remember(body.operationId, payload)
return payload
}
}
return reply.code(500).send({ error: msg })
}
})
app.post('/stop', async (request, reply) => {
const body = request.body as { userId?: string; operationId?: string }
if (body.userId === undefined || body.operationId === undefined) {
return reply.code(400).send({ error: 'userId and operationId are required' })
}
const cached = cache.results.get(body.operationId)
if (cached !== undefined) return cached
const before = await spawner.status(body.userId)
await spawner.stop(body.userId)
if (tunnel !== undefined && tunnelReady && before.main?.port !== undefined) await tunnel.cancel(before.main.port)
epochs.delete(body.userId)
apiKeys.delete(body.userId) // 凭据只该活在实例生命周期内
const payload = { ok: true }
remember(body.operationId, payload)
return payload
})
app.get('/status/:userId', async (request, reply) => {
const { userId } = request.params as { userId: string }
const status = await spawner.status(userId)
return reply.send({ userId, main: status.main ?? null })
})
/** 代理目标(Manager 用)。未运行时返回 `{ running: false }`。 */
app.get('/endpoint/:userId', async (request) => {
const { userId } = request.params as { userId: string }
const endpoint = await spawner.endpointFor(userId)
return endpoint === undefined
? { running: false }
: { running: true, host: options.instanceHost ?? '127.0.0.1', port: endpoint.port }
})
/** self-fencing:我持有的 epoch 落后于 Manager 下发的值 ⇒ **自杀**。 */
app.post('/fence', async (request, reply) => {
const body = request.body as { userId?: string; epoch?: number }
if (body.userId === undefined || body.epoch === undefined) {
return reply.code(400).send({ error: 'userId and epoch are required' })
}
const mine = epochs.get(body.userId)
if (mine === undefined || mine >= body.epoch) return { fenced: false, mine: mine ?? null }
await spawner.stop(body.userId)
epochs.delete(body.userId)
return { fenced: true, mine }
})
app.post('/restart-probe/:userId', async (request) => {
const { userId } = request.params as { userId: string }
return spawner.restartAndProbe(userId)
})
/**
* 活动信号转发(Manager 代理到用户流量时调用)。
* 为什么要转发:idle-reap 是**本地语义**(`LocalSpawner` 的 `lastActive` + TTL/LRU),
* 不转发的话 worker 会以为实例一直没人用、把它回收掉(档案 08)。
*/
app.post('/touch/:userId', async (request) => {
const { userId } = request.params as { userId: string }
spawner.touch(userId)
return { ok: true }
})
app.post('/watchdog/:userId', async (request) => {
const { userId } = request.params as { userId: string }
return { instance: (await spawner.spawnWatchdog(userId)) ?? null }
})
// ── 文件面(T08 S5;供 Manager 的 RemoteUserFs 调用)──────────────────────
// 请求体/响应都是**工作区相对路径 + base64**,与 `UserFs` 的语义一一对应;
// 失败时回 `{error: code}` 并把 `UserFsError.code` 映射成对应 HTTP 状态
// —— 这正是 `user-fs.ts` 里那个 seam 设计的用法(路由按 code 回给前端)。
/** 统一包装:把 `UserFsError` 还原成 wire 形态(其余错误 → 500)。 */
const fsCall = async <T>(reply: FastifyReply, fn: () => Promise<T>): Promise<T | undefined> => {
try {
return await fn()
} catch (err) {
if (err instanceof UserFsError) {
await reply.code(err.status).send({ error: err.code })
return undefined
}
const msg = err instanceof Error ? err.message : String(err)
await reply.code(500).send({ error: 'internal', detail: msg.slice(0, 200) })
return undefined
}
}
app.post('/fs/init', async (request, reply) => {
const body = request.body as { userId?: string; uid?: number }
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
return (await fsCall(reply, async () => {
await userFs.initUserRoot(body.userId as string, body.uid)
return { ok: true }
})) ?? reply
})
app.post('/fs/list', async (request, reply) => {
const body = request.body as { userId?: string; relPath?: string }
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
return (await fsCall(reply, () => userFs.listDir(body.userId as string, body.relPath ?? ''))) ?? reply
})
app.post('/fs/mkdir', async (request, reply) => {
const body = request.body as { userId?: string; relPath?: string }
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
return (await fsCall(reply, async () => {
await userFs.mkdir(body.userId as string, body.relPath ?? '')
return { ok: true }
})) ?? reply
})
app.post('/fs/create', async (request, reply) => {
const body = request.body as { userId?: string; relPath?: string; name?: string; type?: 'file' | 'dir' }
if (body.userId === undefined || body.name === undefined) {
return reply.code(400).send({ error: 'userId and name are required' })
}
// ⚠️ 必须包成对象:`createEntry` 返回的是字符串(净化后的文件名),直接 return 会被
// Fastify 当 text/plain 发出,而调用方(RemoteUserFs)按 JSON 解析 ⇒ 静默 500。
const created = await fsCall(reply, () =>
userFs.createEntry(body.userId as string, body.relPath ?? '', body.name as string, body.type ?? 'file'),
)
return created === undefined ? reply : { name: created }
})
app.post('/fs/upload', async (request, reply) => {
const body = request.body as { userId?: string; relPath?: string; name?: string; dataBase64?: string }
if (body.userId === undefined || body.name === undefined || body.dataBase64 === undefined) {
return reply.code(400).send({ error: 'userId, name and dataBase64 are required' })
}
// 同上:`upload` 返回的是净化后的文件名,必须包成对象。
const uploaded = await fsCall(reply, () =>
userFs.upload(
body.userId as string,
body.relPath ?? '',
body.name as string,
Buffer.from(body.dataBase64 as string, 'base64'),
),
)
return uploaded === undefined ? reply : { name: uploaded }
})
app.post('/fs/read', async (request, reply) => {
const body = request.body as { userId?: string; relPath?: string; maxBytes?: number }
if (body.userId === undefined || body.relPath === undefined) {
return reply.code(400).send({ error: 'userId and relPath are required' })
}
const out = await fsCall(reply, () => userFs.readFile(body.userId as string, body.relPath as string, body.maxBytes))
if (out === undefined) return reply
return { name: out.name, dataBase64: out.data.toString('base64') }
})
app.post('/fs/isdir', async (request, reply) => {
const body = request.body as { userId?: string; relPath?: string }
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
const out = await fsCall(reply, () => userFs.isDirectory(body.userId as string, body.relPath ?? ''))
return out === undefined ? reply : { isDirectory: out }
})
app.post('/fs/plugins', async (request, reply) => {
const body = request.body as { userId?: string }
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
return (await fsCall(reply, () => userFs.listInstalledPlugins(body.userId as string))) ?? reply
})
app.post('/fs/handoff', async (request, reply) => {
const body = request.body as { userId?: string; content?: string }
if (body.userId === undefined || body.content === undefined) {
return reply.code(400).send({ error: 'userId and content are required' })
}
return (await fsCall(reply, async () => {
await userFs.writeHandoff(body.userId as string, body.content as string)
return { ok: true }
})) ?? reply
})
/** 本机 dataRoot(Manager 的 RemoteUserFs 用它做 `resolvePath` 的路径数学)。 */
app.get('/fs/root', async () => ({ dataRoot: config.dataRoot }))
app.get('/', async () => ({ agent: 'dshs-worker', hostId: options.hostId }))
return {
app,
spawner,
/**
* 停机:**先收实例、再关 HTTP**。
* 为什么必须收:实例是 worker 自己的子进程,停机不收就变孤儿(占用端口与内存);
* 而且孤儿会继承 stdout ⇒ 调用方的管道永不关闭(2026-09-15 实测:verify 脚本挂死)。
* 归属与租约由 Manager 侧处理(worker 不写控制面数据),所以这里只停进程。
*/
stop: async (): Promise<void> => {
if (tunnelTimer !== undefined) clearInterval(tunnelTimer)
await app.close()
await spawner.teardown()
await tunnel?.close()
},
}
}
+193
View File
@@ -0,0 +1,193 @@
/**
* Worker 侧的**反向隧道管理器**(T08 跨机演练)。
*
* 为什么需要它:Manager 要连 Worker 上的两样东西 —— **agent 端口**与**每个实例的端口**
* (实例只监听 `127.0.0.1`,这是 portGuard 的设计前提)。而 Worker 公网入方向通常被
* **云安全组**挡住(实测:106 的 19100 从 47 与本机都连不上),放通只能在控制台点。
*
* 绕法:**让 Worker 主动拨 Manager**,用 SSH 反向转发把两边的 `127.0.0.1:<port>` 接起来。
* 好处(实测):
* · **两端都不用新开端口** —— 只用已开放的 SSH 端口(47 是 32022);
* · 链路是加密的,且 Manager 侧落在 loopback(`GatewayPorts no` 默认)⇒ 不对外暴露;
* · 实例端口是**动态**的(`findFreePort()`)⇒ 用 **ControlMaster + `ssh -O forward/cancel`**
* 在**同一条长连接**上加/减转发,不必为每个端口重开连接。
*
* ⚠️ 定位:这是**演练级**传输(生产长期方案见设计 §2.3:受控网段白名单或隧道服务)。
* ⚠️ 默认**关闭**:只有设了 `DSHS_TUNNEL_TARGET` 才启用 ⇒ 对同机/单机形态零影响。
*
* @module dshs/worker/tunnel
*/
import { execFile } from 'node:child_process'
import { existsSync, unlinkSync } from 'node:fs'
import { promisify } from 'node:util'
const run = promisify(execFile)
export interface TunnelOptions {
/** 拨入目标,形如 `[email protected]:32022`。 */
target: string
/** 私钥路径(建议专用、且在 Manager 侧用 `restrict,port-forwarding` 限权)。 */
identity: string
/** ControlMaster socket 路径(同一路径复用同一条连接)。 */
controlPath: string
/** 启动时就转发的端口(agent 自身;还可带控制面 PG 等)。 */
staticPorts?: number[]
/** `ssh` 可执行文件路径。 */
sshBin?: string
}
export class SshTunnel {
/** 内部一律用**已补默认值**的具体类型(否则 `sshBin` 会是 `string | undefined`)。 */
private readonly opts: {
target: string
identity: string
controlPath: string
staticPorts: number[]
sshBin: string
}
private readonly forwarded = new Set<number>()
private readonly hostPart: string
private readonly portPart: number | undefined
constructor(options: TunnelOptions) {
// `user@host:port` 里的 port 是 **SSH 端口**(不是转发的端口)—— 47 上用 32022,
// 必须经 `-p` 传,否则会去连 22 而失败。
const [hostPart, portPart] = options.target.split(':')
this.hostPart = hostPart
this.portPart = portPart === undefined ? undefined : Number(portPart)
this.opts = {
target: options.target,
identity: options.identity,
controlPath: options.controlPath,
staticPorts: options.staticPorts ?? [],
sshBin: options.sshBin ?? '/usr/bin/ssh',
}
}
/** 所有 ssh 调用的公共参数(`-p` 只在目标里显式给了端口时才加)。 */
private baseArgs(): string[] {
return this.portPart === undefined ? [] : ['-p', String(this.portPart)]
}
/** 当前已转发的端口(诊断用)。 */
get ports(): number[] {
return [...this.forwarded]
}
/**
* 建立(或复用)ControlMaster 长连接,并把 `staticPorts` 转发上去。
* 幂等:socket 已存在且 master 还活着就直接返回。
*/
async ensureMaster(): Promise<void> {
if (existsSync(this.opts.controlPath)) {
try {
await run(this.opts.sshBin, [...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'check', this.hostPart])
// master 活着 ⇒ 只需补齐静态转发
for (const port of this.opts.staticPorts ?? []) await this.forward(port)
return
} catch {
try {
unlinkSync(this.opts.controlPath) // 僵尸 socket:清掉重建
} catch {
/* 无所谓 */
}
}
}
const args = [
'-M',
'-N',
'-f',
...this.baseArgs(),
'-S',
this.opts.controlPath,
'-i',
this.opts.identity,
'-o',
'BatchMode=yes',
'-o',
'StrictHostKeyChecking=accept-new',
'-o',
'ExitOnForwardFailure=yes',
'-o',
'ServerAliveInterval=15',
'-o',
'ServerAliveCountMax=4',
]
for (const port of this.opts.staticPorts ?? []) args.push('-R', `${port}:127.0.0.1:${port}`)
args.push(this.hostPart)
await run(this.opts.sshBin, args, { timeout: 20_000 })
for (const port of this.opts.staticPorts ?? []) this.forwarded.add(port)
}
/**
* master 是否还活着(`ssh -O check`)。
*
* 为什么需要:**对端 SSH 重启/断链后,转发会全部消失,而本地 `forwarded` 集合并不知情**
* ⇒ 若只看本地状态,会以为"隧道还好",实际 Manager 已经连不上这台 Worker
* (2026-09-15 收口"以跑通为目的"时补)。
*/
async isMasterAlive(): Promise<boolean> {
try {
await run(this.opts.sshBin, [...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'check', this.hostPart], {
timeout: 8_000,
})
return true
} catch {
// check 失败 ⇒ master 不在了;顺手清掉本地记账,避免"以为还转着"
this.forwarded.clear()
return false
}
}
/**
* 动态加一条反向转发(实例起来时调用)。
* 用**同一个端口号**:实例在 Worker 上是 `127.0.0.1:<port>`,反向转发落到 Manager 的
* `127.0.0.1:<port>` ⇒ Manager 侧无需端口映射表,`endpointFor` 直接回 `127.0.0.1`。
*/
async forward(port: number): Promise<boolean> {
if (this.forwarded.has(port)) return true
try {
await run(
this.opts.sshBin,
[...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'forward', '-R', `${port}:127.0.0.1:${port}`, this.hostPart],
{ timeout: 10_000 },
)
this.forwarded.add(port)
return true
} catch {
return false // 失败不抛:实例本身仍在本机可用,只是跨机代理这跳不可用
}
}
/** 撤销一条转发(实例停止/退出时调用)。 */
async cancel(port: number): Promise<void> {
if (!this.forwarded.has(port)) return
try {
await run(
this.opts.sshBin,
[...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'cancel', '-R', `${port}:127.0.0.1:${port}`, this.hostPart],
{ timeout: 10_000 },
)
} catch {
/* 连接已断也一样算撤销 */
}
this.forwarded.delete(port)
}
/** 关闭 master(进程退出时)。 */
async close(): Promise<void> {
try {
await run(this.opts.sshBin, [...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'exit', this.hostPart], {
timeout: 10_000,
})
} catch {
/* 已退出 */
}
this.forwarded.clear()
}
/** 目标 SSH 端口(`root@h:32022` → 32022)。 */
get targetPort(): number | undefined {
return this.portPart
}
}
+183
View File
@@ -0,0 +1,183 @@
/**
* T08 S2 · 实例归属租约单测。
*
* 刻意**不 mock 时钟**:走真实 SQL(SqliteAdapter(':memory:'))并用**极短 TTL** 制造过期,
* 这样测到的是"SQL 的原子抢占真的成立",而不是"我的假时钟算对了"。
*
* 两个后端都跑(同 T08 S1 的做法):默认 SQLite(内存库,每用例一份);
* 设 `LEASETEST_PG_URL=postgres://…` 时改跑 PG —— 用来验证两套实现语义一致。
* 运行:node --test test/lease.test.mjs
* LEASETEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node --test test/lease.test.mjs
*/
import { test } from 'node:test'
import assert from 'node:assert/strict'
import pg from 'pg'
import { SqliteAdapter } from '../lib/db/sqlite.js'
import { openPgAdapter } from '../lib/db/pg.js'
import { InstanceLease, stillHolder, DEFAULT_LEASE_TTL_MS } from '../lib/supervisor/lease.js'
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms))
const PG_URL = process.env.LEASETEST_PG_URL
/** PG 侧:每个用例前清空三张表(该库专供本测试,清空是安全的)。 */
async function resetPg() {
const client = new pg.Client({ connectionString: PG_URL })
await client.connect()
await client.query('DELETE FROM dsh_instances')
await client.query('DELETE FROM dsh_hosts')
await client.query('DELETE FROM users')
await client.end()
}
/** 每个用例一个独立后端(SQLite = 新内存库;PG = 清空后的专用库)。 */
async function freshDb() {
const db = PG_URL === undefined ? new SqliteAdapter(':memory:', 100000) : await openPgAdapter(PG_URL, 100000)
if (PG_URL !== undefined) await resetPg()
await db.createUser({
id: 'u1',
username: 'alice',
passHash: 'x',
role: 'active',
homeDir: '/tmp/u1',
})
return db
}
/** 短 TTL 的租约(ttl=60ms > 2×20ms,满足不变量)。 */
function shortLease(db, hostId) {
return new InstanceLease(db, hostId, { ttlMs: 60, renewMs: 20 })
}
test('lease: 默认时序满足 ttl > 2×renew 不变量', () => {
assert.ok(DEFAULT_LEASE_TTL_MS > 2 * 10_000, '默认 30s TTL 必须 > 2×10s 续租')
})
test('lease: 违反不变量时构造即抛(fail-loud,防抖动误判)', async () => {
const db = await freshDb()
assert.throws(() => new InstanceLease(db, 'w-a', { ttlMs: 100, renewMs: 60 }), /ttlMs/)
await db.close()
})
test('lease: 首次抢占成功,epoch 从 1 开始', async () => {
const db = await freshDb()
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
const r = await a.acquire('u1')
assert.equal(r.ok, true)
assert.equal(r.epoch, 1)
assert.ok(r.leaseUntil > Date.now(), '租约应在未来')
assert.deepEqual(a.holdings().get('u1'), { epoch: 1, hostId: 'w-a' })
await db.close()
})
test('lease: 未过期时他人抢占失败 —— 单写者保证', async () => {
const db = await freshDb()
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
const b = new InstanceLease(db, 'w-b', { ttlMs: 5000, renewMs: 1000 })
assert.equal((await a.acquire('u1')).ok, true)
const r = await b.acquire('u1')
assert.equal(r.ok, false, '有人在管 ⇒ 必须退让')
assert.equal(r.holder, 'w-a')
assert.equal(b.holdings().has('u1'), false, '失败不得写入本地持有记录')
await db.close()
})
test('lease: 续租必须带 epoch —— 旧持有者续租失败(fencing 生效)', async () => {
const db = await freshDb()
const a = shortLease(db, 'w-a')
assert.equal((await a.acquire('u1')).ok, true)
const staleEpoch = a.holdings().get('u1').epoch
await sleep(90) // 让租约过期
const b = shortLease(db, 'w-b')
const taken = await b.acquire('u1')
assert.equal(taken.ok, true, '过期后可被抢占')
assert.equal(taken.epoch, staleEpoch + 1, 'epoch 必须递增')
// 老持有者拿着旧 epoch 续租 ⇒ 必须失败(否则就脑裂双写了)
assert.equal(await a.renew('u1'), false)
assert.equal(a.holdings().has('u1'), false, '失权后必须清掉本地记录(供 self-fence)')
// 直接调 DB 层也一样:epoch 不匹配 → 不更新
assert.equal(await db.renewInstanceLease('u1', 'w-a', staleEpoch, 60), false)
assert.equal(await b.renew('u1'), true, '新持有者续租成功')
await db.close()
})
test('lease: 释放后归零,可再次抢占且 epoch 继续递增', async () => {
const db = await freshDb()
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
await a.acquire('u1')
assert.equal(await a.release('u1'), true)
assert.equal(a.holdings().has('u1'), false)
const inst = await db.findUserInstance('u1', 'main')
assert.equal(inst.hostId, null, '释放 = 归属清空')
assert.equal(stillHolder(inst, 'w-a', 1), false)
const b = new InstanceLease(db, 'w-b', { ttlMs: 5000, renewMs: 1000 })
const r = await b.acquire('u1')
assert.equal(r.ok, true)
assert.equal(r.epoch, 2, 'epoch 单调递增(不复用)')
await db.close()
})
test('lease: release 也带 epoch 校验 —— 老持有者不能清掉新持有者的归属', async () => {
const db = await freshDb()
const a = shortLease(db, 'w-a')
await a.acquire('u1')
await sleep(90)
const b = shortLease(db, 'w-b')
await b.acquire('u1')
// a 试图释放(它本地已失权 ⇒ release 返回 false 且不动 DB)
assert.equal(await a.release('u1'), false)
const inst = await db.findUserInstance('u1', 'main')
assert.equal(inst.hostId, 'w-b', '新持有者的归属不能被误清')
await db.close()
})
test('lease: stillHolder 判据(hostId + epoch 双匹配)', async () => {
const db = await freshDb()
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
await a.acquire('u1')
const inst = await db.findUserInstance('u1', 'main')
assert.equal(stillHolder(inst, 'w-a', 1), true)
assert.equal(stillHolder(inst, 'w-a', 2), false, 'epoch 不符 = 已失权')
assert.equal(stillHolder(inst, 'w-b', 1), false, '换了机器 = 已失权')
assert.equal(stillHolder(undefined, 'w-a', 1), false)
await db.close()
})
test('lease: 对账视图 —— mine() 只回本机、expiredAll() 回全局过期', async () => {
const db = await freshDb()
await db.createUser({ id: 'u2', username: 'bob', passHash: 'x', role: 'active', homeDir: '/tmp/u2' })
const a = shortLease(db, 'w-a')
const b = shortLease(db, 'w-b')
await a.acquire('u1')
await b.acquire('u2')
assert.deepEqual((await a.mine()).map((i) => i.userId), ['u1'], '一次拿回整机(本机只有 u1)')
assert.deepEqual((await b.mine()).map((i) => i.userId), ['u2'])
assert.equal((await a.expiredAll()).length, 0, '刚认领未过期')
await sleep(90)
assert.equal((await a.expiredAll()).length, 2, '过期后两台都进清单(供巡检,不自动接管)')
assert.equal((await a.expiredHere()).length, 1, 'expiredHere 只回自己名下')
await db.close()
})
test('lease: renewAll 回传失权清单(调用方据此 self-fence)', async () => {
const db = await freshDb()
const a = shortLease(db, 'w-a')
await a.acquire('u1')
assert.deepEqual(await a.renewAll(), [], '正常时无人失权')
await sleep(90)
const b = shortLease(db, 'w-b')
await b.acquire('u1') // 抢走
assert.deepEqual(await a.renewAll(), ['u1'], 'a 必须知道自己已失权')
await db.close()
})