feat(cluster): 集群化落地 —— Manager/Worker 拆分 + 归属租约 + 跨机验证(T08)
背景:把平台从「单机单进程」改造成「1 组 Manager + N 台 Worker + 共享归属状态」,
硬约束 = 全程兼容单例模式(deployMode 默认 local;生产切换前 47 一行未动)。
主要改动
1) 数据模型 v7(SQLite 与 PG 两方言同步):新增 dsh_hosts 注册表 +
dsh_instances.{host_id,epoch,heartbeat_at,lease_until};claimInstance 原子抢占
(UPDATE … WHERE host_id IS NULL OR lease_until < now)+ pinInstanceHost 钉住归属。
2) 租约与 fencing:src/supervisor/lease.ts(acquire/renew/release + stillHolder 判据 +
ttl > 2×renew 硬校验);心跳里续租,失权即向 worker 下发更高 epoch(self-fencing)。
⚠️ release 只清租约(lease_until),**保留 host_id** —— host_id 是「用户数据在哪台」的锚点。
3) Worker agent(src/worker/agent.ts,子命令 dshs worker):实例生命周期 + 文件面 /fs/*
+ 幂等键(operationId)+ 鉴权(timingSafeEqual);Worker 不写控制面数据
(apiKey/uid 由 Manager 随 launch 投递,R5 收窄)。
4) 远端 Spawner + LeasedSpawner:按 host 路由(**粘性优先**:有历史归属且那台 up 就留在原地,
否则按容量选最空的)+ 容量准入 + deployMode=cluster 装配(systemd drop-in,可回滚)。
5) bwrap 修正:**所有挂载点的中间目录统一前置 + 去重 + 由外到内**(「就近创建」会在嵌套前缀下
遮掉已绑挂载点 ⇒ bwrap: Can't chdir);且**只能用 --tmpfs**,用 --perms 会让 47 的
bwrap 0.4.0 直接拒启动(沙箱全挂)。
6) 跨机隧道 src/worker/tunnel.ts:SSH ControlMaster + 动态 -R 转发;**自愈由 agent 本地
20s 定时器驱动**(不能只放 /healthz —— 心跳本身经隧道进来,断了就没人触发它)。
7) 文件面按归属路由(RemoteUserFs):实例与文件必须落在同一台机器,否则实例看不到自己的文件。
8) 观测面:dshs doctor / dshs cluster status。
验证(本次均已实跑)
- test/lease.test.mjs:SQLite 10/10 == PG 10/10
- 组件级端到端 5 个:verify-cluster-{agent,lease,fs,migrate,live}.mjs
- 真跨机(47 Manager / 106 Worker,跨云 + 反向隧道)verify-cluster-cross.mjs 九步全绿
- 域名形态访问 verify-cluster-domain.mjs(<user>.域名 → Manager → 远端实例;越权 403)
- 冒烟 scripts/smoke-*:6/8,失败项与改动前基线完全相同(无回归)
- 生产切换与回滚剧本见 dsh-server-docs/交接单/T08-集群化落地-兼容单例模式.md §16
This commit is contained in:
1 parent
68c0a320ed
commit
c70d5d860e
47 files changed
+5367
-14
No files matched your search
@@ -0,0 +1,11 @@
|
||||
#!/usr/bin/env bash
|
||||
export PGPASSWORD=dshs_cluster_2026
|
||||
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
|
||||
echo "=== users ==="
|
||||
Q "select id || ' | ' || username || ' | ' || role || ' | uid=' || coalesce(uid::text,'-') from users order by row_id"
|
||||
echo "=== dsh_instances ==="
|
||||
Q "select id || ' | user=' || user_id || ' | ' || role || ' | ' || status || ' | host=' || coalesce(host_id,'NULL') || ' | epoch=' || epoch from dsh_instances order by id"
|
||||
echo "=== dsh_hosts ==="
|
||||
Q "select id || ' | cap=' || capacity_mb || ' | used=' || used_mb || ' | ' || status from dsh_hosts order by id"
|
||||
echo "=== sessions(应为 0 条 switch-verify) ==="
|
||||
Q "select count(*) from sessions where user_agent='switch-verify'"
|
||||
@@ -54,5 +54,10 @@ if (role === 'watchdog') {
|
||||
})
|
||||
server.listen(port, '127.0.0.1', () => {
|
||||
console.log(`fake-dsh listening on ${port}`)
|
||||
// 真实 dsh 启动后会打印**可直达的带 token URL**,平台就是靠这行取 launch token
|
||||
// (正则:/dsh web: http://////127//.0//.0//.1://d+/////?token=([A-Za-z0-9_-]+)/)。
|
||||
// 夹具必须照实吐出来,否则平台只能等满 10 s 超时 ⇒ 「登录直达会话」这条链路
|
||||
// 在本机测试里**永远测不到**(2026-09-15 T08 S3 实测踩到)。
|
||||
console.log(`dsh web: http://127.0.0.1:${port}/?token=FAKE_TOKEN_${process.pid}`)
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,186 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* T08 · S1.2:SQLite → Postgres 一次性数据迁移。
|
||||
*
|
||||
* 设计要点(都是踩过才会疼的地方):
|
||||
* 1. **列清单不写死** —— 从 PG 的 information_schema 与 SQLite 的 PRAGMA 取**交集**,
|
||||
* 这样 schema 演进(v4 的 folder/patch、v6 的 enabled 等)不会让脚本静默少搬字段。
|
||||
* 2. **PG 表结构不由本脚本建** —— 先 import 平台自己的 `createDbAdapter`(带 dbUrl),
|
||||
* 让**平台的迁移**在 PG 上建库。这样"迁移脚本"与"平台 schema"永远不会两套。
|
||||
* 3. **identity 列要 `OVERRIDING SYSTEM VALUE`** —— `users.uid` 与 `audit_log.id` 是
|
||||
* GENERATED ALWAYS AS IDENTITY;不覆盖就会重排 id,**uid 一变 = 所有用户文件属主失配**。
|
||||
* 搬完必须 `RESTART WITH` 把序列推到 max+1,否则下一条 INSERT 撞主键。
|
||||
* 4. **FK 顺序**:先 users,再 workspaces/sessions,最后引用它们的表。
|
||||
* 5. `--dry-run` 只报行数,不写任何东西。
|
||||
*
|
||||
* 用法:
|
||||
* node scripts/migrate-sqlite-to-pg.mjs --sqlite /var/lib/dshs/dshs.db \
|
||||
* --pg postgres://dshs:***@127.0.0.1:15432/dshs [--dry-run]
|
||||
*
|
||||
* @module dshs/scripts/migrate-sqlite-to-pg
|
||||
*/
|
||||
import { existsSync } from 'node:fs'
|
||||
import Database from 'better-sqlite3'
|
||||
import pg from 'pg'
|
||||
import { createDbAdapter } from '../lib/db/index.js'
|
||||
import { resolveConfig } from '../lib/config.js'
|
||||
|
||||
/** FK 依赖顺序(父 → 子)。未列出的表会被追加到末尾并告警。 */
|
||||
const ORDER = [
|
||||
'users',
|
||||
'workspaces',
|
||||
'sessions',
|
||||
'folder_plugins',
|
||||
'dsh_instances',
|
||||
'domains',
|
||||
'credential_vault',
|
||||
'business_plugins',
|
||||
'audit_log',
|
||||
]
|
||||
|
||||
/** identity 列(必须 OVERRIDING SYSTEM VALUE + 搬完 RESTART)。 */
|
||||
const IDENTITY = { users: 'uid', audit_log: 'id' }
|
||||
|
||||
/**
|
||||
* ⛔ **绝不搬**的表。
|
||||
*
|
||||
* `schema_migrations`:目标端的"已应用版本"标记由**平台的迁移**建立(见 `ensurePgSchema`),
|
||||
* 从源库搬会把同一批版本号再插一遍 ⇒ `schema_migrations_pkey` 唯一键冲突
|
||||
* (2026-09-15 实测踩到,事务已整体回滚)。语义上也应如此:**结构版本由平台在目标端决定**。
|
||||
*/
|
||||
const SKIP = new Set(['schema_migrations'])
|
||||
|
||||
function arg(name, fallback) {
|
||||
const i = process.argv.indexOf(`--${name}`)
|
||||
return i >= 0 && process.argv[i + 1] !== undefined ? process.argv[i + 1] : fallback
|
||||
}
|
||||
|
||||
const sqlitePath = arg('sqlite')
|
||||
const pgUrl = arg('pg')
|
||||
const dryRun = process.argv.includes('--dry-run')
|
||||
|
||||
if (sqlitePath === undefined || pgUrl === undefined) {
|
||||
console.error('用法: node scripts/migrate-sqlite-to-pg.mjs --sqlite <file> --pg <url> [--dry-run]')
|
||||
process.exit(2)
|
||||
}
|
||||
if (!existsSync(sqlitePath)) {
|
||||
console.error(`SQLite 文件不存在: ${sqlitePath}`)
|
||||
process.exit(2)
|
||||
}
|
||||
|
||||
/** 让**平台的迁移**在 PG 上建好结构(不自己写 DDL,避免两套 schema)。 */
|
||||
async function ensurePgSchema() {
|
||||
const config = resolveConfig({ dataRoot: '/tmp/migrate-tooling', dbUrl: pgUrl })
|
||||
const db = await createDbAdapter(config)
|
||||
await db.close()
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const sq = new Database(sqlitePath, { readonly: true })
|
||||
await ensurePgSchema()
|
||||
const client = new pg.Client({ connectionString: pgUrl })
|
||||
await client.connect()
|
||||
|
||||
const sqTables = sq
|
||||
.prepare("SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'")
|
||||
.all()
|
||||
.map((r) => r.name)
|
||||
const pgTables = (
|
||||
await client.query("SELECT table_name FROM information_schema.tables WHERE table_schema='public'")
|
||||
).rows.map((r) => r.table_name)
|
||||
|
||||
const common = sqTables.filter((t) => pgTables.includes(t) && !SKIP.has(t))
|
||||
if (sqTables.some((t) => SKIP.has(t))) {
|
||||
console.log(`按设计跳过: ${[...SKIP].join(', ')}(目标端的结构版本由平台迁移建立)`)
|
||||
}
|
||||
const ordered = [
|
||||
...ORDER.filter((t) => common.includes(t)),
|
||||
...common.filter((t) => !ORDER.includes(t)),
|
||||
]
|
||||
const extra = common.filter((t) => !ORDER.includes(t))
|
||||
if (extra.length > 0) console.warn(`⚠️ 未在 ORDER 中声明、按末尾处理的表: ${extra.join(', ')}`)
|
||||
|
||||
/** 两端的列交集 —— 只搬双方都有的列。 */
|
||||
async function sharedCols(table) {
|
||||
const sqCols = sq.prepare(`PRAGMA table_info(${table})`).all().map((c) => c.name)
|
||||
const pgCols = (
|
||||
await client.query(
|
||||
'SELECT column_name FROM information_schema.columns WHERE table_schema=$1 AND table_name=$2',
|
||||
['public', table],
|
||||
)
|
||||
).rows.map((r) => r.column_name)
|
||||
return sqCols.filter((c) => pgCols.includes(c))
|
||||
}
|
||||
|
||||
const report = []
|
||||
await client.query('BEGIN')
|
||||
try {
|
||||
for (const table of ordered) {
|
||||
const cols = await sharedCols(table)
|
||||
if (cols.length === 0) {
|
||||
console.warn(`跳过 ${table}: 无公共列`)
|
||||
continue
|
||||
}
|
||||
const rows = sq.prepare(`SELECT ${cols.join(',')} FROM ${table}`).all()
|
||||
const idCol = IDENTITY[table]
|
||||
if (!dryRun && rows.length > 0) {
|
||||
const colList = idCol !== undefined ? [idCol, ...cols.filter((c) => c !== idCol)] : cols
|
||||
const override = idCol !== undefined ? ' OVERRIDING SYSTEM VALUE' : ''
|
||||
const chunk = 200
|
||||
for (let i = 0; i < rows.length; i += chunk) {
|
||||
const slice = rows.slice(i, i + chunk)
|
||||
const values = []
|
||||
const tuples = slice.map((row) => {
|
||||
const ph = colList.map((c) => {
|
||||
values.push(row[c] ?? null)
|
||||
return `$${values.length}`
|
||||
})
|
||||
return `(${ph.join(',')})`
|
||||
})
|
||||
await client.query(
|
||||
`INSERT INTO ${table} (${colList.join(',')})${override} VALUES ${tuples.join(',')}`,
|
||||
values,
|
||||
)
|
||||
}
|
||||
if (idCol !== undefined) {
|
||||
await client.query(
|
||||
`SELECT setval(pg_get_serial_sequence('${table}','${idCol}'),
|
||||
(SELECT COALESCE(MAX(${idCol}),0) FROM ${table}))`,
|
||||
)
|
||||
}
|
||||
}
|
||||
report.push({ table, rows: rows.length, cols: cols.length })
|
||||
}
|
||||
if (dryRun) await client.query('ROLLBACK')
|
||||
else await client.query('COMMIT')
|
||||
} catch (err) {
|
||||
await client.query('ROLLBACK')
|
||||
throw err
|
||||
}
|
||||
|
||||
console.log(`\n${dryRun ? '【DRY-RUN,已回滚】' : '【已提交】'} SQLite → PG 迁移明细`)
|
||||
for (const r of report) console.log(` ${r.table.padEnd(18)} ${String(r.rows).padStart(6)} 行 / ${r.cols} 列`)
|
||||
|
||||
// 校验:逐表比对行数
|
||||
let bad = 0
|
||||
if (!dryRun) {
|
||||
for (const r of report) {
|
||||
const pgCount = Number((await client.query(`SELECT COUNT(*) AS c FROM ${r.table}`)).rows[0].c)
|
||||
const sqCount = Number(sq.prepare(`SELECT COUNT(*) AS c FROM ${r.table}`).get().c)
|
||||
if (pgCount !== sqCount) {
|
||||
console.error(` ✗ 行数不符 ${r.table}: PG=${pgCount} SQLite=${sqCount}`)
|
||||
bad += 1
|
||||
}
|
||||
}
|
||||
console.log(bad === 0 ? '\n✅ 逐表行数一致' : `\n❌ ${bad} 张表行数不符`)
|
||||
}
|
||||
|
||||
await client.end()
|
||||
sq.close()
|
||||
process.exit(bad === 0 ? 0 : 1)
|
||||
}
|
||||
|
||||
main().catch((err) => {
|
||||
console.error(err)
|
||||
process.exit(1)
|
||||
})
|
||||
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env bash
|
||||
# 存量数据并行搬运 47 → 106(4 路并发;瓶颈在源端小文件 IOPS,单流只 ~0.4MB/s)
|
||||
# 搬完写 /root/push-parallel.done,供后续 cutover 判断
|
||||
set -uo pipefail
|
||||
KEY=/root/.ssh/dshworker_ed25519
|
||||
DST=[email protected]
|
||||
SSHO="-i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o Compression=no"
|
||||
SRC=/var/lib/dshs
|
||||
LOG=/root/push-parallel.log
|
||||
: > "$LOG"
|
||||
|
||||
ssh -n $SSHO "$DST" 'mkdir -p /var/lib/dshs'
|
||||
echo "[$(date +%T)] 目标端就绪" >> "$LOG"
|
||||
|
||||
# 按"大小"分组:大用户单开一路,其余合流(每路一个 tar→ssh)
|
||||
one() { # $1=组名 其余=相对路径列表
|
||||
local name="$1"; shift
|
||||
(
|
||||
cd "$SRC" || exit 1
|
||||
{ for p in "$@"; do [ -e "$p" ] && echo "$p"; done; } > "/tmp/list-$name.txt"
|
||||
tar --numeric-owner --files-from="/tmp/list-$name.txt" -cf - \
|
||||
| ssh $SSHO "$DST" "tar -C /var/lib/dshs --numeric-owner -xf -"
|
||||
echo "[$(date +%T)] $name 完成 rc=$?" >> "$LOG"
|
||||
) &
|
||||
}
|
||||
|
||||
# 组划分(按实际内容:1 个大用户 + 若干小目录)
|
||||
one g1 "users/4092b965-2f68-4977-9989-68b3966f7df0"
|
||||
one g2 "users/cce6d1cd-b376-4304-80f0-0e1c58c9ffde" "users/3ec95f69-6a4e-4d16-a415-56aa09396fc5"
|
||||
one g3 "users/4eaeb26b-9e0c-4c68-9b61-0daf70664ae5" "users/74e8804a-e0c8-4645-84de-90dd3fae6c2b"
|
||||
one g4 "users/7ba268be-6103-438c-8e6b-609b422ccbca" "users/ca3f36e0-937f-437e-b850-53b8230f20f8" \
|
||||
"bundled-skills" "business-plugins" "whitelist-cache"
|
||||
|
||||
wait
|
||||
echo "[$(date +%T)] 全部完成" >> "$LOG"
|
||||
ssh -n $SSHO "$DST" 'du -sm /var/lib/dshs | cut -f1' >> "$LOG" 2>&1
|
||||
touch /root/push-parallel.done
|
||||
@@ -0,0 +1,63 @@
|
||||
#!/usr/bin/env bash
|
||||
# 在 47 初始化「控制面 PG」:独立数据目录 /var/lib/dshs-pg + 专用 unit dshs-pg.service
|
||||
# 设计口径:控制面 DB 在 Manager 侧、仅 loopback、scram 认证。
|
||||
set -uo pipefail
|
||||
PGDATA=/var/lib/dshs-pg
|
||||
PGPORT=15432
|
||||
DBPW="${DSHS_PG_PASSWORD:-dshs_cluster_2026}"
|
||||
PGVER=$(/usr/bin/postgres --version | grep -oE '[0-9]+' | head -1)
|
||||
echo " PG 版本: $(/usr/bin/postgres --version)"
|
||||
|
||||
# 别让发行版的默认单元意外起来(我们用自己的 unit + 自己的数据目录)
|
||||
systemctl disable --now postgresql 2>/dev/null >/dev/null || true
|
||||
|
||||
if [ ! -f "$PGDATA/PG_VERSION" ]; then
|
||||
install -d -o postgres -g postgres -m 700 "$PGDATA"
|
||||
su - postgres -c "/usr/bin/initdb -D $PGDATA -E UTF8 --locale=C.UTF-8 --auth-local=peer --auth-host=scram-sha-256" >/tmp/initdb.log 2>&1 \
|
||||
&& echo " ✓ initdb 完成" || { echo " ✗ initdb 失败"; tail -5 /tmp/initdb.log; exit 1; }
|
||||
cat >> "$PGDATA/postgresql.conf" <<CONF
|
||||
|
||||
# ── DSHS 控制面(2026-09-15 集群化切换)──
|
||||
listen_addresses = '127.0.0.1'
|
||||
port = $PGPORT
|
||||
unix_socket_directories = '/var/run/postgresql'
|
||||
max_connections = 100
|
||||
shared_buffers = 128MB
|
||||
CONF
|
||||
chown postgres:postgres "$PGDATA/postgresql.conf"
|
||||
echo " ✓ 已写入 listen=127.0.0.1 port=$PGPORT"
|
||||
fi
|
||||
|
||||
cat > /etc/systemd/system/dshs-pg.service <<UNIT
|
||||
[Unit]
|
||||
Description=DSHS control-plane PostgreSQL (cluster mode)
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=notify
|
||||
User=postgres
|
||||
Group=postgres
|
||||
ExecStart=/usr/bin/postgres -D $PGDATA
|
||||
ExecReload=/bin/kill -HUP \$MAINPID
|
||||
KillMode=mixed
|
||||
TimeoutStopSec=30
|
||||
Restart=on-failure
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
UNIT
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl enable --now dshs-pg >/dev/null 2>&1
|
||||
sleep 3
|
||||
echo " dshs-pg: $(systemctl is-active dshs-pg) 监听: $(ss -lntp 2>/dev/null | grep -c $PGPORT)"
|
||||
|
||||
# 角色 + 库(幂等)
|
||||
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_roles where rolname='dshs'\"" 2>/dev/null | grep -q 1 \
|
||||
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create role dshs login password '$DBPW'\"" >/dev/null 2>&1
|
||||
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_database where datname='dshs'\"" 2>/dev/null | grep -q 1 \
|
||||
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create database dshs owner dshs\"" >/dev/null 2>&1
|
||||
|
||||
echo " --- 连接自检(dshs 角色) ---"
|
||||
PGPASSWORD="$DBPW" /usr/bin/psql -h 127.0.0.1 -p $PGPORT -U dshs -d dshs -tAc "select current_user||'@'||current_database()||' pg='||version()" 2>&1 | head -1 | cut -c1-90
|
||||
echo " 连接串(含密码,勿外传): postgres://dshs:$DBPW@127.0.0.1:$PGPORT/dshs"
|
||||
@@ -0,0 +1,16 @@
|
||||
#!/usr/bin/env bash
|
||||
# 在 47 的控制面 PG 上建角色与库(peer 认证走 unix socket,不依赖已存在的密码)
|
||||
set -uo pipefail
|
||||
PGPORT=15432
|
||||
DBPW="dshs_cluster_2026"
|
||||
|
||||
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_roles where rolname='dshs'\"" 2>/dev/null | grep -q 1 \
|
||||
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create role dshs login password '$DBPW'\"" >/dev/null 2>&1
|
||||
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select 1 from pg_database where datname='dshs'\"" 2>/dev/null | grep -q 1 \
|
||||
|| su - postgres -c "/usr/bin/psql -p $PGPORT -c \"create database dshs owner dshs\"" >/dev/null 2>&1
|
||||
|
||||
echo "=== 角色/库确认(本地 peer) ==="
|
||||
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select rolname from pg_roles where rolname='dshs'\"" 2>/dev/null | sed 's/^/ role: /'
|
||||
su - postgres -c "/usr/bin/psql -p $PGPORT -tAc \"select datname||' owner='||pg_get_userbyid(datdba) from pg_database where datname='dshs'\"" 2>/dev/null | sed 's/^/ db: /'
|
||||
echo "=== 连接自检(TCP + 密码,走的应是**本机 PG 13**) ==="
|
||||
PGPASSWORD="$DBPW" /usr/bin/psql -h 127.0.0.1 -p $PGPORT -U dshs -d dshs -tAc "select version()" 2>&1 | head -1 | cut -c1-80
|
||||
@@ -75,7 +75,11 @@ try {
|
||||
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
|
||||
console.log('launch ->', r.status, r.body?.url)
|
||||
assert(r.status === 200, 'launch succeeds')
|
||||
assert(r.body.url === 'https://carol.test.local/', 'launch returns the subdomain URL')
|
||||
// 2026-09-15(T08 S3):夹具 `fake-dsh.mjs` 现在**照实打印带 token 的 URL**(与真实 dsh 一致),
|
||||
// 于是这里不能再写死成不带 token 的相等 —— 原断言是"夹具不吐 token"时的意外产物。
|
||||
// 保留原意(是子域 URL、不泄露回环端口),并把 token 的存在一并纳入判据。
|
||||
assert(r.body.url.startsWith('https://carol.test.local/'), 'launch returns the subdomain URL')
|
||||
assert(!r.body.url.includes('127.0.0.1'), 'launch URL must not leak the loopback port')
|
||||
|
||||
await sleep(200)
|
||||
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
#!/usr/bin/env bash
|
||||
# 起一台 **cluster 模式的 Manager**(T08 跨机演练用;在 Manager 那台机器上跑)。
|
||||
#
|
||||
# 现场的对照(2026-09-15 演练实测):
|
||||
# · Manager 在 **47**(本脚本所在机器),监听 `127.0.0.1:13080`(**不公网暴露**)
|
||||
# · Worker agent 在 **106**,经 SSH 反向隧道出现在本机 `127.0.0.1:19000` / `19001`
|
||||
# · 控制面 PG 也在 **106**,经同一条隧道出现在本机 `127.0.0.1:15432`
|
||||
# 用法:bash scripts/start-cluster-manager.sh (env 见下方 manager.env)
|
||||
在 47 上:建 manager.env、bootstrap 管理员、起 cluster Manager(127.0.0.1:13080)
|
||||
set -uo pipefail
|
||||
cd /opt/dshs-cluster || exit 1
|
||||
|
||||
cat > /opt/dshs-cluster/manager.env <<'ENVEOF'
|
||||
DSHS_DEPLOY_MODE=cluster
|
||||
DSHS_DB_URL=postgres://dshs:[email protected]:15432/dshs_cross
|
||||
DSHS_DATA_ROOT=/opt/dshs-cluster/data
|
||||
DSHS_CLUSTER_HOST_ID=m-47
|
||||
DSHS_CLUSTER_AGENT_URL=http://127.0.0.1:19000
|
||||
DSHS_CLUSTER_AGENT_TOKEN=cross-machine-token
|
||||
DSHS_CLUSTER_INSTANCE_HOST=127.0.0.1
|
||||
DSHS_CLUSTER_WORKER_DATA_ROOT=/opt/dshs-cluster/live-data
|
||||
DSHS_CLUSTER_CAPACITY_MB=-1
|
||||
DSHS_CLUSTER_REGISTER_SELF=0
|
||||
DSHS_CLUSTER_LEASE_TTL_MS=30000
|
||||
ENVEOF
|
||||
|
||||
set -a
|
||||
# shellcheck disable=SC1091
|
||||
. /opt/dshs-cluster/manager.env
|
||||
set +a
|
||||
|
||||
echo "--- bootstrap 管理员 ---"
|
||||
node lib/cli.js bootstrap-admin --username root --password crossmgr123 2>&1 | tail -1
|
||||
|
||||
echo "--- 起 Manager ---"
|
||||
pkill -f "dshs-cluster/lib/cli.js --port 13080" 2>/dev/null
|
||||
sleep 1
|
||||
nohup node lib/cli.js --port 13080 --host 127.0.0.1 --log-level warn > /tmp/manager-47.log 2>&1 &
|
||||
sleep 7
|
||||
|
||||
echo "--- 自检 ---"
|
||||
echo " login.html : $(curl -s -o /dev/null -w '%{http_code}' -m 6 http://127.0.0.1:13080/login.html)"
|
||||
echo " 进程 : $(pgrep -cf 'dshs-cluster/lib/cli.js --port 13080')"
|
||||
echo " 日志尾部 :"
|
||||
tail -4 /tmp/manager-47.log 2>/dev/null | sed 's/^/ /'
|
||||
@@ -0,0 +1,76 @@
|
||||
#!/usr/bin/env bash
|
||||
# 切换 A 步(在 47 上跑):
|
||||
# ① 备份 /opt/dshs/lib → /opt/dsh/backups/lib-<ts>/
|
||||
# ② 覆盖 /opt/dshs/lib(T08 集群版代码)
|
||||
# ③ 装 **本地 Worker**(w-47,19100,无隧道 —— Manager 同机直连)
|
||||
# ④ 把既有用户(admin/guest)的归属**预置**为 w-47(否则粘性落点无处可粘、新老用户会被按容量随机调度)
|
||||
# ⑤ 在 PG 里注册 w-47 / w-106 两台 worker
|
||||
set -uo pipefail
|
||||
TS=$(date +%Y%m%d-%H%M%S)
|
||||
W47_TOKEN="dshs-worker-47-c4b7e19f"
|
||||
W106_TOKEN="dshs-worker-7f3a91c05e"
|
||||
PGURL="postgres://dshs:[email protected]:15432/dshs"
|
||||
TARBALL=/tmp/dshs-lib-new.tgz
|
||||
|
||||
echo "=== ① 备份 /opt/dshs/lib ==="
|
||||
mkdir -p "/opt/dsh/backups/lib-$TS"
|
||||
cp -a /opt/dshs/lib "/opt/dsh/backups/lib-$TS/lib" && echo " ✓ 备份到 /opt/dsh/backups/lib-$TS/lib($(find /opt/dsh/backups/lib-$TS -type f | wc -l) 文件)"
|
||||
|
||||
echo "=== ② 覆盖 lib ==="
|
||||
[ -f "$TARBALL" ] || { echo " ✗ 缺少 $TARBALL"; exit 1; }
|
||||
rm -rf /opt/dshs/lib && tar -xzf "$TARBALL" -C /opt/dshs
|
||||
echo " ✓ 已覆盖;cluster 特征检查: $(grep -l "DEPLOY_MODE" /opt/dshs/lib/config.js >/dev/null 2>&1 && echo '有 cluster 代码 ✓' || echo '✗ 未见 cluster 代码')"
|
||||
echo " lease/agent/tunnel: $(ls /opt/dshs/lib/supervisor/lease.js /opt/dshs/lib/worker/agent.js /opt/dshs/lib/worker/tunnel.js 2>/dev/null | wc -l)/3"
|
||||
|
||||
echo "=== ③ 本地 Worker 单元(w-47,无隧道) ==="
|
||||
cat > /etc/dshs-worker.env <<ENV
|
||||
DSHS_DATA_ROOT=/var/lib/dshs
|
||||
DSHS_ISOLATION_MODE=account
|
||||
DSHS_DSH_BIN=/usr/local/bin/dsh
|
||||
DSHS_BASE_UID=100000
|
||||
DSH_INSTANCE_NODE_OPTIONS=--max-old-space-size=160
|
||||
DSH_INSTANCE_UNIVER_SOCKET=auto
|
||||
DSHS_CLUSTER_AGENT_TOKEN=$W47_TOKEN
|
||||
ENV
|
||||
chmod 600 /etc/dshs-worker.env
|
||||
cat > /etc/systemd/system/dshs-worker.service <<UNIT
|
||||
[Unit]
|
||||
Description=DSHS cluster worker agent (this host = 47, local users' instances)
|
||||
After=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
EnvironmentFile=/etc/dshs-worker.env
|
||||
ExecStart=/usr/local/bin/node /opt/dshs/lib/cli.js worker --port 19100 --host 127.0.0.1 --host-id w-47 --instance-host 127.0.0.1 --log-level info
|
||||
Restart=on-failure
|
||||
RestartSec=3
|
||||
KillMode=mixed
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
UNIT
|
||||
systemctl daemon-reload; systemctl enable dshs-worker >/dev/null 2>&1
|
||||
systemctl restart dshs-worker; sleep 5
|
||||
echo " dshs-worker=$(systemctl is-active dshs-worker) healthz=$(curl -s -m 6 http://127.0.0.1:19100/healthz | head -c 120)"
|
||||
|
||||
echo "=== ④ 既有用户归属预置为 w-47(粘性锚点) ==="
|
||||
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
|
||||
"insert into dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
|
||||
select 'dsh-'||id, id, 'main', 'stopped', 'w-47', 0, 0, 0 from users
|
||||
on conflict (id) do update set host_id='w-47', lease_until=0" 2>&1 | tail -1
|
||||
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
|
||||
"select u.username||' -> '||coalesce(i.host_id,'NULL') from users u left join dsh_instances i on i.user_id=u.id" 2>&1 | sed 's/^/ /'
|
||||
|
||||
echo "=== ⑤ 注册两台 worker ==="
|
||||
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
|
||||
"insert into dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status)
|
||||
values ('w-47','http://127.0.0.1:19100','$W47_TOKEN',1024,0,'up'),
|
||||
('w-106','http://127.0.0.1:19000','$W106_TOKEN',2560,0,'up')
|
||||
on conflict (id) do update set endpoint=excluded.endpoint, agent_token=excluded.agent_token, capacity_mb=excluded.capacity_mb, status='up'" 2>&1 | tail -1
|
||||
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
|
||||
"select id||' cap='||capacity_mb||' status='||status||' ep='||endpoint from dsh_hosts order by id" 2>&1 | sed 's/^/ /'
|
||||
|
||||
echo "=== 回滚剧本(现在就记下) ==="
|
||||
echo " rm -f /etc/systemd/system/dshs.service.d/cluster.conf && systemctl daemon-reload && \\"
|
||||
echo " systemctl restart dshs # 回 SQLite 单机;lib 回滚 = cp -a /opt/dsh/backups/lib-$TS/lib /opt/dshs/lib"
|
||||
echo " 备份时间戳: $TS"
|
||||
@@ -0,0 +1,20 @@
|
||||
#!/usr/bin/env bash
|
||||
# A 步:把 47 的生产库 SQLite → 本机控制面 PG(停机窗口内做,避免迁移期间库变动)
|
||||
set -uo pipefail
|
||||
PGURL="postgres://dshs:[email protected]:15432/dshs"
|
||||
DB=/var/lib/dshs/dshs.db
|
||||
cd /opt/dshs-cluster || exit 1
|
||||
|
||||
echo "=== 1) 停 dshs(窗口开始) ==="
|
||||
systemctl stop dshs
|
||||
sleep 2
|
||||
echo " dshs=$(systemctl is-active dshs) 实例 scope 残留: $(systemctl list-units --type=scope --all 2>/dev/null | grep -c 'dsh-' || echo 0)"
|
||||
echo " 库文件: $(ls -l $DB $DB-wal 2>/dev/null | awk '{print $5}' | tr '\n' '/')"
|
||||
|
||||
echo "=== 2) dry-run(只报行数) ==="
|
||||
node scripts/migrate-sqlite-to-pg.mjs --sqlite "$DB" --pg "$PGURL" --dry-run 2>&1 | tail -18
|
||||
|
||||
echo "=== 3) 真迁 ==="
|
||||
node scripts/migrate-sqlite-to-pg.mjs --sqlite "$DB" --pg "$PGURL" 2>&1 | tail -18
|
||||
echo " rc=$?"
|
||||
echo " ⏸ 窗口保持关闭 —— 部署 worker 与 drop-in 后再一起开(见后续步骤)"
|
||||
@@ -0,0 +1,21 @@
|
||||
#!/usr/bin/env bash
|
||||
# 校验:SQLite 与 PG 两侧的 users 明细 + 与磁盘目录对照(判断 2 vs 7 是孤儿目录还是迁移漏行)
|
||||
set -uo pipefail
|
||||
cd /opt/dshs-cluster || exit 1
|
||||
echo "=== SQLite 侧(只读打开,含 WAL) ==="
|
||||
node -e '
|
||||
const D = require("better-sqlite3");
|
||||
const db = new D("/var/lib/dshs/dshs.db", { readonly: true });
|
||||
const users = db.prepare("select id, username, role, uid from users order by rowid").all();
|
||||
console.log(" users 行数:", users.length);
|
||||
for (const u of users) console.log(` ${u.username} role=${u.role} uid=${u.uid} id=${u.id}`);
|
||||
console.log(" credential_vault:", db.prepare("select count(*) c from credential_vault").get().c);
|
||||
console.log(" business_plugins:", db.prepare("select count(*) c from business_plugins").get().c);
|
||||
'
|
||||
echo "=== PG 侧 ==="
|
||||
PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
|
||||
"select username||' role='||role||' uid='||coalesce(uid::text,'NULL')||' id='||id from users order by rowid" 2>&1 | sed 's/^/ /'
|
||||
echo " PG users 总数: $(PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc 'select count(*) from users')"
|
||||
echo "=== 磁盘目录 vs DB ==="
|
||||
echo " 目录($(ls /var/lib/dshs/users | wc -l) 个):"
|
||||
ls /var/lib/dshs/users | sed 's/^/ /'
|
||||
@@ -0,0 +1,22 @@
|
||||
#!/usr/bin/env bash
|
||||
# B1 步(在 47 上跑):uid 保真校验 + 建 47→106 专用密钥并打印公钥
|
||||
set -uo pipefail
|
||||
KEY=/root/.ssh/dshworker_ed25519
|
||||
|
||||
echo "=== 1) uid 保真校验(PG 与 SQLite 必须一致 —— 否则 106 上文件属主全错) ==="
|
||||
echo -n " PG : "; PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc \
|
||||
"select string_agg(username||'='||coalesce(uid::text,'NULL'), ' ' order by row_id) from users" 2>&1 | head -1
|
||||
echo -n " SQLite : "; node -e 'const D=require("better-sqlite3");const db=new D("/var/lib/dshs/dshs.db",{readonly:true});console.log(db.prepare("select username, uid from users order by rowid").all().map(u=>u.username+"="+u.uid).join(" "))'
|
||||
echo " --- 磁盘目录属主(与 uid 对照;多出的 5 个是已删用户孤儿目录) ---"
|
||||
for d in /var/lib/dshs/users/*/; do printf " %-38s uid=%s\n" "$(basename "$d")" "$(stat -c %u "$d")"; done
|
||||
|
||||
echo "=== 2) 建 47→106 专用密钥(仅用于 rsync) ==="
|
||||
[ -f "$KEY" ] || ssh-keygen -t ed25519 -N "" -C "dshs-rsync-47to106" -f "$KEY" >/dev/null 2>&1
|
||||
echo " PUBKEY=$(cat "$KEY.pub")"
|
||||
|
||||
echo "=== 3) 试连通 106:22 ==="
|
||||
if ssh -i "$KEY" -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=8 [email protected] 'echo ok' 2>/dev/null | grep -q ok; then
|
||||
echo " ✓ 已可连通(公钥已装)"
|
||||
else
|
||||
echo " ⏳ 尚不可连通 —— 需先把我本机把这个公钥装到 106"
|
||||
fi
|
||||
@@ -0,0 +1,38 @@
|
||||
#!/usr/bin/env bash
|
||||
# 切换 B 步(在 47 上跑):加 systemd drop-in → 重启 dshs(= 切换时刻,约 1-2 秒中断)
|
||||
# · 用 drop-in 而非改 unit:unit 本体 hash 不变 ⇒ 回滚只需删 drop-in
|
||||
# · 同时把 106 的 worker lib 也更新到同一版本(两侧代码必须一致)
|
||||
set -uo pipefail
|
||||
mkdir -p /etc/systemd/system/dshs.service.d
|
||||
cat > /etc/systemd/system/dshs.service.d/cluster.conf <<CONF
|
||||
# T08 集群化(2026-09-15):Manager 在 47、实例落在 w-47(既有用户)/ w-106(新用户)
|
||||
# 回滚:删除本文件 → systemctl daemon-reload → systemctl restart dshs
|
||||
[Service]
|
||||
Environment="DSHS_DEPLOY_MODE=cluster"
|
||||
Environment="DSHS_DB_URL=postgres://dshs:[email protected]:15432/dshs"
|
||||
Environment="DSHS_CLUSTER_HOST_ID=w-47"
|
||||
Environment="DSHS_CLUSTER_AGENT_URL=http://127.0.0.1:19100"
|
||||
Environment="DSHS_CLUSTER_AGENT_TOKEN=dshs-worker-47-c4b7e19f"
|
||||
Environment="DSHS_CLUSTER_INSTANCE_HOST=127.0.0.1"
|
||||
Environment="DSHS_CLUSTER_WORKER_DATA_ROOT=/var/lib/dshs"
|
||||
Environment="DSHS_CLUSTER_CAPACITY_MB=-1"
|
||||
Environment="DSHS_CLUSTER_REGISTER_SELF=0"
|
||||
Environment="DSHS_CLUSTER_LEASE_TTL_MS=30000"
|
||||
CONF
|
||||
echo " ✓ drop-in 已写($(wc -l < /etc/systemd/system/dshs.service.d/cluster.conf) 行)"
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl restart dshs
|
||||
sleep 6
|
||||
|
||||
echo "=== 切换后自检 ==="
|
||||
echo " dshs=$(systemctl is-active dshs) | dshs-pg=$(systemctl is-active dshs-pg) | dshs-worker=$(systemctl is-active dshs-worker)"
|
||||
echo " 生效 env(systemd 解析后):"
|
||||
systemctl show dshs -p Environment 2>/dev/null | tr ' ' '\n' | grep -E "DEPLOY_MODE|CLUSTER_HOST_ID|CLUSTER_AGENT_URL|DB_URL" | sed 's/^/ /'
|
||||
echo " 门户: login.html=$(curl -s -o /dev/null -w '%{http_code}' -m 8 http://127.0.0.1:3080/login.html)"
|
||||
echo " 公网: https://alotbuy.com/login.html = $(curl -s -o /dev/null -w '%{http_code}' -m 12 https://alotbuy.com/login.html)"
|
||||
echo " --- dshs cluster status ---"
|
||||
cd /opt/dshs && set -a && . /etc/dshs.env && set +a && set -a && . /etc/systemd/system/dshs.service.d/cluster.conf 2>/dev/null || true
|
||||
cd /opt/dshs && DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" node lib/cli.js cluster status 2>&1 | head -8 | sed 's/^/ /'
|
||||
echo " --- 回滚命令(随时可用) ---"
|
||||
echo " rm -f /etc/systemd/system/dshs.service.d/cluster.conf && systemctl daemon-reload && systemctl restart dshs"
|
||||
@@ -0,0 +1,33 @@
|
||||
#!/usr/bin/env bash
|
||||
# B2 步(在 47 上跑):rsync 生产 dataRoot → 106
|
||||
# · --numeric-ids 保 uid/gid(否则 106 上文件属主全错 ⇒ 实例 EACCES)
|
||||
# · 排除 dshs.db*(DB 权威源已是 47 的 PG)与 secret.key(凭据主密钥不外扩到 Worker —— R5 最小面)
|
||||
# · ⚠️ 所有 ssh 调用带 -n:脚本本身经 stdin 传入,ssh 若不隔离 stdin 会把**脚本剩余部分**吃掉
|
||||
set -uo pipefail
|
||||
KEY=/root/.ssh/dshworker_ed25519
|
||||
DST=[email protected]
|
||||
SRC=/var/lib/dshs
|
||||
SSHOPT="-n -i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=10"
|
||||
|
||||
command -v rsync >/dev/null || { dnf -y install rsync >/tmp/dnf-rsync.log 2>&1 && echo " ✓ 47 装上 rsync"; }
|
||||
|
||||
echo "=== 连通性 + 对端 rsync ==="
|
||||
ssh $SSHOPT "$DST" 'command -v rsync >/dev/null || dnf -y install rsync >/tmp/dnf-rs.log 2>&1; echo " 对端: $(rsync --version | head -1)"' 2>&1 | tail -2
|
||||
|
||||
echo "=== 目标端准备 ==="
|
||||
ssh $SSHOPT "$DST" 'mkdir -p /var/lib/dshs && ls -ld /var/lib/dshs' 2>&1 | sed 's/^/ /'
|
||||
|
||||
echo "=== rsync ==="
|
||||
rsync -a --numeric-ids --stats -e "ssh $SSHOPT" \
|
||||
--exclude 'dshs.db' --exclude 'dshs.db-shm' --exclude 'dshs.db-wal' --exclude 'secret.key' \
|
||||
"$SRC/" "$DST:/var/lib/dshs/" 2>&1 | grep -E "Number of regular files transferred|Total file size|sent [0-9]|total size is" | sed 's/^/ /'
|
||||
|
||||
echo "=== 目标端核对 ==="
|
||||
ssh $SSHOPT "$DST" 'bash -c "
|
||||
echo \" 顶层: \$(ls /var/lib/dshs | tr \"\n\" \" \")\"
|
||||
echo \" users 目录数: \$(ls /var/lib/dshs/users 2>/dev/null | wc -l)\"
|
||||
for d in /var/lib/dshs/users/*/; do printf \" %-38s uid=%s\n\" \"\$(basename \$d)\" \"\$(stat -c %u \$d)\"; done
|
||||
echo \" bundled-skills: \$(ls /var/lib/dshs/bundled-skills 2>/dev/null | wc -l) 项\"
|
||||
echo \" business-plugins: \$(ls /var/lib/dshs/business-plugins 2>/dev/null | wc -l) 项\"
|
||||
echo \" ⛔ 不应存在(dshs.db/secret.key): \$(ls /var/lib/dshs/dshs.db /var/lib/dshs/secret.key 2>/dev/null | wc -l) 个(应为 0)\"
|
||||
"' 2>&1
|
||||
@@ -0,0 +1,30 @@
|
||||
#!/usr/bin/env bash
|
||||
# B2' 步:47 → 106 直推生产 dataRoot(tar-over-ssh;只用命令执行,绕开 rsync 协议问题)
|
||||
# · tar --numeric-owner 保 uid/gid(否则 106 上属主错 ⇒ 实例 EACCES)
|
||||
# · 排除 dshs.db*(权威源=47 的 PG)与 secret.key(主密钥不外扩 —— R5 最小面)
|
||||
set -uo pipefail
|
||||
KEY=/root/.ssh/dshworker_ed25519
|
||||
DST=[email protected]
|
||||
SSHO="-i $KEY -o BatchMode=yes -o StrictHostKeyChecking=accept-new"
|
||||
|
||||
echo "=== 1) 目标端准备(ssh -n 防抢 stdin) ==="
|
||||
ssh -n $SSHO "$DST" 'mkdir -p /var/lib/dshs && echo " ready: $(ls -ld /var/lib/dshs)"'
|
||||
|
||||
echo "=== 2) 推送(tar → ssh → tar --numeric-owner -x) ==="
|
||||
cd /var/lib/dshs
|
||||
tar --numeric-owner -cf - \
|
||||
--exclude=./dshs.db --exclude=./dshs.db-shm --exclude=./dshs.db-wal --exclude=./secret.key \
|
||||
. | ssh $SSHO "$DST" 'tar -C /var/lib/dshs --numeric-owner -xf -'
|
||||
rc=$?
|
||||
echo " 管道 rc=$rc"
|
||||
|
||||
echo "=== 3) 目标端核对 ==="
|
||||
ssh -n $SSHO "$DST" 'bash -c "
|
||||
echo \" 顶层: \$(ls /var/lib/dshs | tr \"\n\" \" \")\"
|
||||
echo \" 总量: \$(du -sh /var/lib/dshs | cut -f1)\"
|
||||
echo \" users 目录数: \$(ls /var/lib/dshs/users | wc -l)\"
|
||||
echo \" --- 属主抽样(应与 47 的 uid 一致) ---\"
|
||||
for d in /var/lib/dshs/users/*/; do printf \" %-38s uid=%s\n\" \"\$(basename \$d)\" \"\$(stat -c %u \$d)\"; done
|
||||
echo \" bundled-skills=\$(ls /var/lib/dshs/bundled-skills | wc -l) business-plugins=\$(ls /var/lib/dshs/business-plugins | wc -l)\"
|
||||
echo \" ⛔ 不该有 dshs.db/secret.key: \$(ls /var/lib/dshs/dshs.db /var/lib/dshs/secret.key 2>/dev/null | wc -l) 个(应 0)\"
|
||||
"'
|
||||
@@ -0,0 +1,79 @@
|
||||
#!/usr/bin/env bash
|
||||
# 最终验证(在 47 上跑):清污染 → 重置锚点 → 重验两条路径
|
||||
# ① 既有用户 guest:留 w-47 + 工作区有历史数据 + 实例页正常
|
||||
# ② 新用户:落 w-106 + **文件真的写到 106 的盘** + 实例页正常
|
||||
set -uo pipefail
|
||||
export PGPASSWORD=dshs_cluster_2026
|
||||
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
|
||||
M=http://127.0.0.1:3080
|
||||
DOMAIN=alotbuy.com
|
||||
T47=dshs-worker-47-c4b7e19f
|
||||
T106=dshs-worker-7f3a91c05e
|
||||
SSH106="ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected]"
|
||||
|
||||
mksess() {
|
||||
local u="$1" T H
|
||||
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
|
||||
Q "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
|
||||
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+1800000, '127.0.0.1','switch-verify' from users where username='$u'" >/dev/null
|
||||
printf '%s' "$T"
|
||||
}
|
||||
owner() { Q "select coalesce(i.host_id,'NULL')||' epoch='||coalesce(i.epoch,0) from users u left join dsh_instances i on i.user_id=u.id where u.username='$1'"; }
|
||||
agent() { curl -s -m 6 -H "x-dsh-agent-token: $2" "http://127.0.0.1:$1/healthz" | grep -o '"instances":[0-9]*'; }
|
||||
isapp() { grep -q '<base href=' <<<"$1" && echo "✓实例页" || echo "✗非实例页"; }
|
||||
|
||||
echo "########## 0) 清污染:停所有实例 + 删测试用户 ##########"
|
||||
systemctl restart dshs-worker; sleep 4
|
||||
$SSH106 'systemctl restart dshs-worker' >/dev/null 2>&1; sleep 4
|
||||
echo " 重启后 w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
|
||||
AT=$(mksess admin); AC="sid=$AT"
|
||||
for u in $(Q "select id from users where username like 'switchtest%' or username like 'swtest%'"); do
|
||||
echo " 删测试用户 $u → $(curl -s -m 20 -X DELETE -b "$AC" "$M/api/admin/users/$u" -o /dev/null -w '%{http_code}')"
|
||||
done
|
||||
echo " 剩余用户: $(Q "select string_agg(username,', ') from users")"
|
||||
|
||||
echo "########## 1) 重置 guest 锚点(host_id=w-47) ##########"
|
||||
Q "update dsh_instances set host_id='w-47', epoch=0, lease_until=0, status='stopped'
|
||||
where user_id=(select id from users where username='guest')" >/dev/null
|
||||
echo " $(owner guest)"
|
||||
|
||||
echo
|
||||
echo "########## 2) 路径①:既有用户 guest ##########"
|
||||
GT=$(mksess guest); GC="sid=$GT"
|
||||
E=$(curl -s -m 60 -X POST -b "$GC" -H 'content-type: application/json' -d '{}' "$M/api/dsh/enter")
|
||||
sleep 3
|
||||
echo " 归属: $(owner guest) ← 期望 w-47"
|
||||
echo " w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
|
||||
echo " 工作区条目: $(curl -s -m 10 -b "$GC" "$M/api/desktop/tree" | grep -o '"name":"[^"]*"' | head -4 | tr '\n' ' ')"
|
||||
PAGE=$(curl -s -m 25 -L -b "$GC" -H "Host: guest.$DOMAIN" "$M/" | head -c 300)
|
||||
echo " 实例页: $(isapp "$PAGE")"
|
||||
|
||||
echo
|
||||
echo "########## 3) 路径②:新用户 ##########"
|
||||
NU="swtest2$(date +%H%M%S)"
|
||||
echo " register=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" "$M/api/auth/register" -o /dev/null -w '%{http_code}')"
|
||||
NID=$(Q "select id from users where username='$NU'")
|
||||
echo " approve=$(curl -s -m 15 -X POST -b "$AC" "$M/api/admin/users/$NID/approve" -o /dev/null -w '%{http_code}')"
|
||||
NC=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" -D - "$M/api/auth/login" -o /dev/null | grep -i '^set-cookie' | head -1 | grep -oP 'sid=[^;]+')
|
||||
echo " mkdir=$(curl -s -m 15 -X POST -b "$NC" -H 'content-type: application/json' -d '{"path":"proj"}' "$M/api/fs/mkdir" -o /dev/null -w '%{http_code}') ← 首次触达应把归属钉住"
|
||||
echo " 钉住后归属: $(owner "$NU") ← 期望 w-106(与下面的 launch 必须同台)"
|
||||
echo " upload=$(curl -s -m 20 -X POST -b "$NC" -H 'content-type: application/json' -d "{\"path\":\"proj\",\"name\":\"hello.txt\",\"data\":\"$(printf 'hi-from-switch' | base64 -w0)\"}" "$M/api/fs/upload" -o /dev/null -w '%{http_code}')"
|
||||
echo " launch=$(curl -s -m 60 -X POST -b "$NC" -H 'content-type: application/json' -d '{"folder":"proj"}' "$M/api/dsh/launch" -o /dev/null -w '%{http_code}')"
|
||||
sleep 4
|
||||
echo " 归属: $(owner "$NU") ← 期望 w-106(粘性保持)"
|
||||
echo " w-106=$(agent 19000 $T106) w-47=$(agent 19100 $T47)"
|
||||
echo " --- 落盘取证 ---"
|
||||
L47=$($SSH106 "ls /var/lib/dshs/users/$NID/ws/proj/hello.txt 2>/dev/null" 2>/dev/null || true)
|
||||
echo " 106 盘: ${L47:-不存在}"
|
||||
echo " 47 盘: $(ls /var/lib/dshs/users/$NID/ws/proj/hello.txt 2>/dev/null || echo 不存在(应不存在 ✓))"
|
||||
NP=$(curl -s -m 25 -L -b "$NC" -H "Host: $NU.$DOMAIN" "$M/" | head -c 300)
|
||||
echo " 实例页(Host: $NU.$DOMAIN): $(isapp "$NP")"
|
||||
|
||||
echo
|
||||
echo "########## 收尾 ##########"
|
||||
curl -s -m 40 -X POST -b "$GC" "$M/api/dsh/stop" -o /dev/null -w " guest stop=%{http_code}\n"
|
||||
curl -s -m 40 -X POST -b "$NC" "$M/api/dsh/stop" -o /dev/null -w " newuser stop=%{http_code}\n"
|
||||
echo " stop 后 guest 归属(**不应再被清空**): $(owner guest)"
|
||||
Q "delete from sessions where user_agent='switch-verify'" >/dev/null
|
||||
echo " 残留临时 session: $(Q "select count(*) from sessions where user_agent='switch-verify'")(应 0)"
|
||||
echo " 新用户待清: $NU / $NID"
|
||||
@@ -0,0 +1,63 @@
|
||||
#!/usr/bin/env bash
|
||||
# 切换后功能验证 v2(在 47 上跑)
|
||||
# ① 既有用户 guest:应留 w-47,且**工作区有历史数据**(文件在 47 的盘上)
|
||||
# ② 新用户:应落 w-106,且**建的文件真的出现在 106 的盘上**(文件面路由已修)
|
||||
# 判据:实例页必须带 <base href="/"(门户页不算);归属看 PG;文件落盘看两台机器磁盘
|
||||
set -uo pipefail
|
||||
PG() { PGPASSWORD=dshs_cluster_2026 /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
|
||||
M=http://127.0.0.1:3080
|
||||
DOMAIN=alotbuy.com
|
||||
T47=dshs-worker-47-c4b7e19f
|
||||
T106=dshs-worker-7f3a91c05e
|
||||
|
||||
mksess() {
|
||||
local u="$1" T H
|
||||
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
|
||||
PG "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
|
||||
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+1800000, '127.0.0.1','switch-verify' from users where username='$u'" >/dev/null
|
||||
printf '%s' "$T"
|
||||
}
|
||||
owner() { PG "select u.username||' -> '||coalesce(i.host_id,'NULL')||' epoch='||i.epoch from users u left join dsh_instances i on i.user_id=u.id where u.username='$1'"; }
|
||||
agent() { curl -s -m 6 -H "x-dsh-agent-token: $2" "http://127.0.0.1:$1/healthz" | grep -o '"instances":[0-9]*'; }
|
||||
isapp() { grep -q '<base href=' <<<"$1" && echo "✓实例页" || echo "✗非实例页"; }
|
||||
|
||||
echo "############ ① 既有用户 guest(应留 w-47 + 工作区有数据) ############"
|
||||
GT=$(mksess guest); GC="sid=$GT"
|
||||
E=$(curl -s -m 60 -X POST -b "$GC" -H 'content-type: application/json' -d '{}' "$M/api/dsh/enter")
|
||||
echo " enter → $(head -c 130 <<<"$E")"
|
||||
sleep 3
|
||||
echo " 归属: $(owner guest) ← 期望 w-47"
|
||||
echo " w-47=$(agent 19100 $T47) w-106=$(agent 19000 $T106)"
|
||||
echo " 工作区首项: $(curl -s -m 10 -b "$GC" "$M/api/desktop/tree" | head -c 150)"
|
||||
URL=$(grep -o '"url":"[^"]*"' <<<"$E" | head -1 | cut -d'"' -f4)
|
||||
PAGE=$(curl -s -m 25 -L -b "$GC" -H "Host: guest.$DOMAIN" "$M/" | head -c 300)
|
||||
echo " 实例页: $(isapp "$PAGE")"
|
||||
echo " 47 盘上 guest 工作区条目: $(ls /var/lib/dshs/users/4092b965-2f68-4977-9989-68b3966f7df0/ws 2>/dev/null | wc -l) 项(>0 = 数据在 47 ✓)"
|
||||
|
||||
echo
|
||||
echo "############ ② 新用户(应落 w-106 + 文件真的写到 106 盘) ############"
|
||||
NU="swtest$(date +%H%M%S)"
|
||||
AT=$(mksess admin); AC="sid=$AT"
|
||||
echo " register=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" "$M/api/auth/register" -o /dev/null -w '%{http_code}')"
|
||||
NU_ID=$(PG "select id from users where username='$NU'")
|
||||
echo " approve=$(curl -s -m 15 -X POST -b "$AC" "$M/api/admin/users/$NU_ID/approve" -o /dev/null -w '%{http_code}')"
|
||||
NC=$(curl -s -m 15 -X POST -H 'content-type: application/json' -d "{\"username\":\"$NU\",\"password\":\"SwitchTest123\"}" -D - "$M/api/auth/login" -o /dev/null | grep -i '^set-cookie' | head -1 | grep -oP 'sid=[^;]+')
|
||||
echo " mkdir=$(curl -s -m 15 -X POST -b "$NC" -H 'content-type: application/json' -d '{"path":"proj"}' "$M/api/fs/mkdir" -o /dev/null -w '%{http_code}')"
|
||||
echo " upload=$(curl -s -m 20 -X POST -b "$NC" -H 'content-type: application/json' -d "{\"path\":\"proj\",\"name\":\"hello.txt\",\"data\":\"$(printf 'hi-from-switch' | base64 -w0)\"}" "$M/api/fs/upload" -o /dev/null -w '%{http_code}')"
|
||||
echo " launch=$(curl -s -m 60 -X POST -b "$NC" -H 'content-type: application/json' -d '{"folder":"proj"}' "$M/api/dsh/launch" -o /dev/null -w '%{http_code}')"
|
||||
sleep 4
|
||||
echo " 归属: $(owner "$NU") ← 期望 w-106"
|
||||
echo " w-106=$(agent 19000 $T106) w-47=$(agent 19100 $T47)"
|
||||
echo " --- 文件落盘取证(这才是文件面路由修好的证据) ---"
|
||||
echo " 47 盘: $(ls /var/lib/dshs/users/$NU_ID/ws/proj/hello.txt 2>/dev/null && echo 存在 || echo '不存在 ✓(不应在 47)')"
|
||||
echo " 106 盘: $(ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] "ls /var/lib/dshs/users/$NU_ID/ws/proj/hello.txt 2>/dev/null" 2>/dev/null && echo 存在✓ || echo '不存在 ✗')"
|
||||
NU_PAGE=$(curl -s -m 25 -L -b "$NC" -H "Host: $NU.$DOMAIN" "$M/" | head -c 300)
|
||||
echo " 实例页(Host: $NU.$DOMAIN): $(isapp "$NU_PAGE")"
|
||||
|
||||
echo
|
||||
echo "############ 收尾 ############"
|
||||
curl -s -m 40 -X POST -b "$GC" "$M/api/dsh/stop" -o /dev/null -w " guest stop=%{http_code}\n"
|
||||
curl -s -m 40 -X POST -b "$NC" "$M/api/dsh/stop" -o /dev/null -w " newuser stop=%{http_code}\n"
|
||||
PG "delete from sessions where user_agent='switch-verify'" >/dev/null
|
||||
echo " 残留临时 session: $(PG "select count(*) from sessions where user_agent='switch-verify'")(应 0)"
|
||||
echo " 新用户记录: $NU / $NU_ID"
|
||||
@@ -0,0 +1,53 @@
|
||||
#!/usr/bin/env bash
|
||||
# C 步(在 106 上跑):把生产 Worker agent 装成 systemd 单元
|
||||
# · env 与 47 的生产实例侧对齐(DSH_INSTANCE_* 必须一致 —— 实例是在 Worker 上 spawn 的)
|
||||
# · token 走 env 文件(600)而不是命令行,避免 ps 泄露
|
||||
# · 反向隧道复用演练时那把 key(其公钥已在 47 的 authorized_keys 里,restrict,port-forwarding)
|
||||
set -uo pipefail
|
||||
TOKEN="${WORKER_TOKEN:-dshs-worker-7f3a91c05e}"
|
||||
|
||||
cat > /etc/dshs-worker.env <<ENV
|
||||
# DSHS 集群 Worker(2026-09-15 切换)—— 与 47 /etc/dshs.env 的**实例侧**条目保持一致
|
||||
DSHS_DATA_ROOT=/var/lib/dshs
|
||||
DSHS_ISOLATION_MODE=account
|
||||
DSHS_DSH_BIN=/usr/bin/dsh
|
||||
DSHS_BASE_UID=100000
|
||||
DSH_INSTANCE_NODE_OPTIONS=--max-old-space-size=160
|
||||
DSH_INSTANCE_UNIVER_SOCKET=auto
|
||||
# 控制通道:Worker 主动拨 47 的反向隧道(公网入方向被云安全组挡住 ⇒ 只能这个方向)
|
||||
DSHS_CLUSTER_AGENT_TOKEN=$TOKEN
|
||||
[email protected]:32022
|
||||
DSHS_TUNNEL_IDENTITY=/root/.ssh/tunnel_ed25519
|
||||
ENV
|
||||
chmod 600 /etc/dshs-worker.env
|
||||
|
||||
cat > /etc/systemd/system/dshs-worker.service <<UNIT
|
||||
[Unit]
|
||||
Description=DSHS cluster worker agent (hosts per-user dsh instances on 106)
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
EnvironmentFile=/etc/dshs-worker.env
|
||||
ExecStart=/usr/bin/node /opt/dshs-cluster/lib/cli.js worker --port 19000 --host 127.0.0.1 --host-id w-106 --instance-host 127.0.0.1 --log-level info
|
||||
Restart=on-failure
|
||||
RestartSec=3
|
||||
KillMode=mixed
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
UNIT
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl enable dshs-worker >/dev/null 2>&1
|
||||
echo " 单元已装并 enable;token 长度=${#TOKEN}"
|
||||
|
||||
# 旧的手工 agent 若在跑先停(按端口定位)
|
||||
pid=$(ss -lntpH 'sport = :19000' 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
|
||||
[ -n "$pid" ] && kill "$pid" && sleep 2 && echo " 已停旧的手工 agent pid=$pid"
|
||||
|
||||
systemctl restart dshs-worker
|
||||
sleep 6
|
||||
echo " dshs-worker=$(systemctl is-active dshs-worker)"
|
||||
echo " healthz: $(curl -s -m 6 http://127.0.0.1:19000/healthz | head -c 200)"
|
||||
@@ -0,0 +1,36 @@
|
||||
#!/usr/bin/env bash
|
||||
# 切换收尾:清理验证残留 + 健康检查(在 47 上跑)
|
||||
set -uo pipefail
|
||||
export PGPASSWORD=dshs_cluster_2026
|
||||
Q() { /usr/bin/psql -h 127.0.0.1 -p 15432 -U dshs -d dshs -tAc "$1"; }
|
||||
M=http://127.0.0.1:3080
|
||||
|
||||
echo "=== 1) 删掉验证留下的测试用户 ==="
|
||||
T=$(openssl rand -hex 32); H=$(printf '%s' "$T" | sha256sum | cut -d' ' -f1)
|
||||
Q "insert into sessions (token_hash,user_id,created_at,expires_at,ip,user_agent)
|
||||
select '$H', id, (extract(epoch from now())*1000)::bigint, (extract(epoch from now())*1000)::bigint+600000, '127.0.0.1','switch-cleanup' from users where username='admin'" >/dev/null
|
||||
for u in $(Q "select id from users where username like 'swtest%' or username like 'switchtest%'"); do
|
||||
echo " delete $u → $(curl -s -m 20 -X DELETE -b "sid=$T" "$M/api/admin/users/$u" -o /dev/null -w '%{http_code}')"
|
||||
done
|
||||
Q "delete from sessions where user_agent='switch-cleanup'" >/dev/null
|
||||
echo " 剩余用户: $(Q "select string_agg(username||'('||role||')', ', ') from users")"
|
||||
|
||||
echo "=== 2) 停掉验证期间起的实例 ==="
|
||||
systemctl restart dshs-worker; sleep 4
|
||||
ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] 'systemctl restart dshs-worker' >/dev/null 2>&1
|
||||
sleep 4
|
||||
echo " w-47=$(curl -s -m 6 -H 'x-dsh-agent-token: dshs-worker-47-c4b7e19f' http://127.0.0.1:19100/healthz | grep -o '\"instances\":[0-9]*')"
|
||||
echo " w-106=$(ssh -n -i /root/.ssh/dshworker_ed25519 -o BatchMode=yes -o StrictHostKeyChecking=accept-new [email protected] 'curl -s -m 6 -H "x-dsh-agent-token: dshs-worker-7f3a91c05e" http://127.0.0.1:19000/healthz | grep -o .instances.:[0-9]*' 2>/dev/null)"
|
||||
|
||||
echo "=== 3) 健康检查 ==="
|
||||
echo " dshs=$(systemctl is-active dshs) dshs-pg=$(systemctl is-active dshs-pg) dshs-worker=$(systemctl is-active dshs-worker)"
|
||||
echo " 门户公网: $(curl -s -o /dev/null -w '%{http_code}' -m 12 https://alotbuy.com/login.html)"
|
||||
echo " --- dshs cluster status ---"
|
||||
cd /opt/dshs && set -a && . /etc/dshs.env && set +a
|
||||
DSHS_DEPLOY_MODE=cluster DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" \
|
||||
node lib/cli.js cluster status 2>&1 | head -9 | sed 's/^/ /'
|
||||
echo " --- dshs doctor ---"
|
||||
DSHS_DEPLOY_MODE=cluster DSHS_DB_URL="postgres://dshs:[email protected]:15432/dshs" \
|
||||
node lib/cli.js doctor 2>&1 | grep -cE "^✓" | sed 's/^/ ✓ 项数: /'
|
||||
echo " --- 实例归属总览 ---"
|
||||
Q "select u.username || ' → ' || coalesce(i.host_id,'(未指派)') || ' status=' || coalesce(i.status,'-') from users u left join dsh_instances i on i.user_id=u.id order by u.row_id" | sed 's/^/ /'
|
||||
@@ -0,0 +1,31 @@
|
||||
#!/usr/bin/env bash
|
||||
# 拆除演练环境(为生产切换让路):
|
||||
# · 47:演练 Manager(127.0.0.1:13080)
|
||||
# · 106:两个演练 agent(19000/19001)⇒ 会 teardown 实例并关闭隧道
|
||||
# 目的:① 释放 47 的 15432(隧道转发占用 → 生产 PG 要用)
|
||||
# ② 避免"演练 Manager + 生产 Manager 抢同一个 agent"
|
||||
set -uo pipefail
|
||||
|
||||
echo "=== 47 侧:停演练 Manager(按端口定位,绝不碰生产 3080) ==="
|
||||
PID=$(ss -lntpH 'sport = :13080' 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
|
||||
if [ -n "$PID" ]; then kill "$PID" && echo " 已停演练 Manager pid=$PID"; else echo " 13080 无监听"; fi
|
||||
sleep 2
|
||||
|
||||
echo "=== 106 侧:停两个演练 agent ==="
|
||||
for port in 19000 19001; do
|
||||
pid=$(ss -lntpH "sport = :$port" 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
|
||||
[ -n "$pid" ] && kill "$pid" && echo " 已停 agent($port) pid=$pid" || echo " $port 无监听"
|
||||
done
|
||||
sleep 4
|
||||
echo " 残留 dsh 实例: $(pgrep -cf 'dsh --profile' || echo 0)"
|
||||
echo " 隧道进程: $(pgrep -cf 'tunnel_ed25519' || echo 0)"
|
||||
|
||||
echo "=== 47 侧:15432 是否已释放(隧道转发应已消失) ==="
|
||||
ss -lntp 2>/dev/null | grep 15432 || echo " ✓ 15432 已空闲"
|
||||
echo "=== 加固:若隧道 sshd 残留,按端口收掉 ==="
|
||||
for port in 19000 19001 15432; do
|
||||
pid=$(ss -lntpH "sport = :$port" 2>/dev/null | grep -oP 'pid=\K[0-9]+' | head -1)
|
||||
[ -n "$pid" ] && kill "$pid" 2>/dev/null && echo " 收掉残留监听 $port pid=$pid"
|
||||
done
|
||||
sleep 2
|
||||
ss -lntp 2>/dev/null | grep -E "15432|1900[01]" || echo " ✓ 三个端口都已空闲"
|
||||
@@ -0,0 +1,195 @@
|
||||
/**
|
||||
* T08 S3 · 端到端验证:**Manager 经 RemoteSpawner 把实例起在 worker agent 上**。
|
||||
*
|
||||
* 与 `smoke-dsh.mjs` 的区别:那条走的是"本机直接 spawn",这条**多了一跳 HTTP**
|
||||
* (Manager → agent → LocalSpawner),因此它验证的是 S3 真正的交付物:
|
||||
* ① 路由/代理层**一行没改**就能工作(`Spawner` 抽象 + `endpointFor` 的 host:port);
|
||||
* ② **launch token 回传**(P0-6)—— 否则"登录直达会话"与 401 自愈会失效;
|
||||
* ③ **幂等键**:同一 operationId 重发不会起第二个实例(Manager 超时重试是常态);
|
||||
* ④ **self-fencing**:`/fence` 下发的 epoch 更高时,agent 主动停掉自己那个实例。
|
||||
*
|
||||
* 刻意用 **soft 隔离 + stand-in fake-dsh**:本测试要验的是**跨机协议**,
|
||||
* 不是沙箱(沙箱另有 S1.6 的双机证据)。用 account 模式反而会被"夹具路径必须在
|
||||
* 沙箱绑定集内"这条夹具限制干扰(见 `交接单/T08-§10.4`)。
|
||||
*
|
||||
* 运行:node scripts/verify-cluster-agent.mjs
|
||||
*/
|
||||
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { dirname, join } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
import { buildServer } from '../lib/web/server.js'
|
||||
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
|
||||
import { resolveConfig } from '../lib/config.js'
|
||||
import { hashPassword } from '../lib/web/auth.js'
|
||||
|
||||
function assert(condition, message) {
|
||||
if (!condition) throw new Error('ASSERT: ' + message)
|
||||
}
|
||||
|
||||
const here = dirname(fileURLToPath(import.meta.url))
|
||||
const fakeDsh = join(here, 'fake-dsh.mjs')
|
||||
const TOKEN = 'verify-cluster-agent-token'
|
||||
const dataRoot = mkdtempSync(join(tmpdir(), 'dsh-cluster-'))
|
||||
let agentApp
|
||||
let agentHandle
|
||||
let app
|
||||
|
||||
try {
|
||||
// ── 1) 起 worker agent(进程内,端口随机)──────────────────────────────
|
||||
const agentConfig = resolveConfig({
|
||||
port: 0,
|
||||
dbPath: ':memory:',
|
||||
dataRoot,
|
||||
dshCommand: [process.execPath, fakeDsh],
|
||||
clusterHostId: 'w-1',
|
||||
})
|
||||
const agent = buildWorkerAgent(agentConfig, {
|
||||
hostId: 'w-1',
|
||||
token: TOKEN,
|
||||
port: 0,
|
||||
host: '127.0.0.1',
|
||||
instanceHost: '127.0.0.1',
|
||||
logLevel: 'warn',
|
||||
})
|
||||
agentApp = agent.app
|
||||
agentHandle = agent // 收尾要用 agent.stop()(会 teardown 本机实例),否则子进程孤儿化
|
||||
await agentApp.listen({ host: '127.0.0.1', port: 0 })
|
||||
const agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
|
||||
console.log('agent ->', agentUrl)
|
||||
|
||||
const agentJson = async (path, { method = 'GET', body } = {}) => {
|
||||
const res = await fetch(agentUrl + path, {
|
||||
method,
|
||||
headers: {
|
||||
[AGENT_TOKEN_HEADER]: TOKEN,
|
||||
...(body ? { 'content-type': 'application/json' } : {}),
|
||||
},
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
})
|
||||
const text = await res.text()
|
||||
return { status: res.status, body: text ? JSON.parse(text) : null }
|
||||
}
|
||||
|
||||
// agent 存活 + 鉴权(不带 token 必须 401)
|
||||
const hz = await agentJson('/healthz')
|
||||
assert(hz.status === 200 && hz.body.hostId === 'w-1', 'agent healthz')
|
||||
const noAuth = await fetch(agentUrl + '/instances')
|
||||
assert(noAuth.status === 401, 'agent 拒绝无凭据请求')
|
||||
|
||||
// ── 2) 起 Manager(deployMode=cluster → RemoteSpawner)─────────────────
|
||||
const managerConfig = resolveConfig({
|
||||
port: 0,
|
||||
dbPath: ':memory:',
|
||||
dataRoot,
|
||||
deployMode: 'cluster',
|
||||
clusterAgentUrl: agentUrl,
|
||||
clusterAgentToken: TOKEN,
|
||||
clusterInstanceHost: '127.0.0.1',
|
||||
})
|
||||
app = await buildServer(managerConfig)
|
||||
await app.listen({ port: 0 })
|
||||
const base = `http://127.0.0.1:${app.server.address().port}`
|
||||
console.log('manager ->', base, '(deployMode=cluster)')
|
||||
|
||||
await app.db.createUser({
|
||||
id: 'u1',
|
||||
username: 'carol',
|
||||
passHash: await hashPassword('carolpass123'),
|
||||
role: 'active',
|
||||
homeDir: '/tmp/u1-home',
|
||||
})
|
||||
mkdirSync(join(dataRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
|
||||
|
||||
const json = async (path, { method = 'GET', body, cookie } = {}) => {
|
||||
const res = await fetch(base + path, {
|
||||
method,
|
||||
headers: {
|
||||
...(body ? { 'content-type': 'application/json' } : {}),
|
||||
...(cookie ? { cookie } : {}),
|
||||
},
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
})
|
||||
const text = await res.text()
|
||||
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
|
||||
}
|
||||
|
||||
// ── 3) 登录 → 拉起(实例实际落在 agent 上)────────────────────────────
|
||||
let r = await json('/api/auth/login', { method: 'POST', body: { username: 'carol', password: 'carolpass123' } })
|
||||
assert(r.status === 200, 'login succeeds')
|
||||
const cookie = r.setCookie.split(';')[0]
|
||||
|
||||
r = await json('/api/dsh/status', { cookie })
|
||||
assert(r.body.running === false, 'not running initially')
|
||||
|
||||
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
|
||||
console.log('launch ->', r.status, r.body?.url ? 'url 已返回' : r.body)
|
||||
assert(r.status === 200, 'launch succeeds')
|
||||
// ② launch token 回传(P0-6):URL 里必须带 token,否则"登录直达"失效
|
||||
assert(typeof r.body.url === 'string' && r.body.url.includes('token='), 'launch token 必须回传到 URL')
|
||||
|
||||
// 实例真的在 **worker** 上(而不是 Manager 本机)
|
||||
const onAgent = await agentJson('/instances')
|
||||
assert(onAgent.body.instances.length === 1, 'worker 上有 1 个实例')
|
||||
assert(onAgent.body.instances[0].userId === 'u1', 'worker 上的实例属于 u1')
|
||||
console.log('agent 视角 -> 实例数', onAgent.body.instances.length)
|
||||
|
||||
r = await json('/api/dsh/status', { cookie })
|
||||
assert(r.body.running === true, 'running after launch')
|
||||
|
||||
// ── 4) 代理链路(endpointFor → agent 给的 host:port)──────────────────
|
||||
let proxyText
|
||||
for (let i = 0; i < 20; i += 1) {
|
||||
try {
|
||||
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie } })
|
||||
if (res.status === 200) {
|
||||
proxyText = await res.text()
|
||||
break
|
||||
}
|
||||
} catch {
|
||||
/* 子进程还没监听,重试 */
|
||||
}
|
||||
await new Promise((resolve) => setTimeout(resolve, 100))
|
||||
}
|
||||
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), 'proxy reaches the child DSH(经远端协议)')
|
||||
console.log('proxy -> 200 且命中 fake-dsh')
|
||||
|
||||
// ── 5) 幂等键:同 operationId 重发不得起第二个实例 ─────────────────────
|
||||
const opId = 'verify-idempotent-1'
|
||||
const l1 = await agentJson('/launch', { method: 'POST', body: { userId: 'u1', folder: join(dataRoot, 'users', 'u1', 'ws', 'proj'), patch: undefined, operationId: opId } })
|
||||
const l2 = await agentJson('/launch', { method: 'POST', body: { userId: 'u1', folder: join(dataRoot, 'users', 'u1', 'ws', 'proj'), patch: undefined, operationId: opId } })
|
||||
assert(l1.status === 200 && l2.status === 200, '重复 launch 不报错')
|
||||
const afterIdem = await agentJson('/instances')
|
||||
assert(afterIdem.body.instances.length === 1, '幂等:仍然只有 1 个实例')
|
||||
console.log('幂等 -> 同 operationId 重发后实例数仍为', afterIdem.body.instances.length)
|
||||
|
||||
// ── 6) self-fencing:更高 epoch 下发 ⇒ agent 主动停掉自己那个实例 ───────
|
||||
await agentJson('/launch', { method: 'POST', body: { userId: 'u1', epoch: 1, operationId: 'verify-epoch-1' } })
|
||||
const f1 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 1 } })
|
||||
assert(f1.body.fenced === false, 'epoch 相同 ⇒ 不被 fence')
|
||||
const f2 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 2 } })
|
||||
assert(f2.body.fenced === true, 'epoch 更高 ⇒ self-fence')
|
||||
const afterFence = await agentJson('/instances')
|
||||
assert(afterFence.body.instances.length === 0, 'fence 后实例已停')
|
||||
console.log('self-fence -> epoch 1→2 触发,实例已停止')
|
||||
|
||||
// ── 7) 停止 ───────────────────────────────────────────────────────────
|
||||
r = await json('/api/dsh/stop', { method: 'POST', cookie })
|
||||
assert(r.status === 200, 'stop succeeds')
|
||||
r = await json('/api/dsh/status', { cookie })
|
||||
assert(r.body.running === false, 'stopped after stop')
|
||||
|
||||
console.log('\nOK: cluster 模式(Manager → worker agent → 实例)端到端通过')
|
||||
console.log(' ✓ 路由/代理层零改动 ✓ launch token 回传 ✓ 幂等键 ✓ self-fencing')
|
||||
} finally {
|
||||
await app?.close()
|
||||
// ⚠️ 必须走 agent.stop():它先 teardown 本机实例再关 HTTP —— 否则 fake-dsh 孤儿会继承
|
||||
// stdout,管道不关 ⇒ ssh / CI 挂死(2026-09-15 实测)。
|
||||
await agentHandle?.stop()
|
||||
await new Promise((resolve) => setTimeout(resolve, 500))
|
||||
try {
|
||||
rmSync(dataRoot, { recursive: true, force: true })
|
||||
} catch {
|
||||
// best-effort:Windows 上子进程的 cwd 还在里面时会 EBUSY(temp 目录会被系统回收)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,184 @@
|
||||
/**
|
||||
* T08 · **真跨机演练**驱动脚本(在 Manager 那台机器上运行)。
|
||||
*
|
||||
* 与 `verify-cluster-live.mjs`(同机、脚本自己起进程)的区别:这里**假设两侧都已部署好**:
|
||||
* · Manager 运行在**本机**(47)`http://127.0.0.1:13080`
|
||||
* · Worker agent 运行在**另一台机器**(106),经 **SSH 反向隧道**出现在本机 `127.0.0.1:19000`
|
||||
* · 控制面 PG 也在**另一台机器**(106)上,经隧道出现在本机 `127.0.0.1:15432`
|
||||
* 它回答的是本次演练的核心问题:**跨机到底能不能用**(含跨机代理取页面、跨 worker 迁移)。
|
||||
*
|
||||
* 运行(在 47 上):MANAGER=http://127.0.0.1:13080 AGENT_TOKEN=cross-machine-token \
|
||||
* AGENT2=http://127.0.0.1:19001 node scripts/verify-cluster-cross.mjs
|
||||
*/
|
||||
const MANAGER = process.env.MANAGER ?? 'http://127.0.0.1:13080'
|
||||
const TOKEN = process.env.AGENT_TOKEN ?? 'cross-machine-token'
|
||||
const AGENT1 = process.env.AGENT1 ?? 'http://127.0.0.1:19000'
|
||||
const AGENT2 = process.env.AGENT2 ?? ''
|
||||
const ADMIN_PW = process.env.ADMIN_PW ?? 'crossmgr123'
|
||||
const USER_PW = 'crossuser123'
|
||||
|
||||
function assert(condition, message) {
|
||||
if (!condition) throw new Error('ASSERT: ' + message)
|
||||
}
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
||||
|
||||
const json = async (path, { method = 'GET', body, cookie } = {}) => {
|
||||
const res = await fetch(MANAGER + path, {
|
||||
method,
|
||||
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
})
|
||||
const text = await res.text()
|
||||
let parsed = null
|
||||
try {
|
||||
parsed = text === '' ? null : JSON.parse(text)
|
||||
} catch {
|
||||
parsed = { raw: text.slice(0, 200) }
|
||||
}
|
||||
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
|
||||
}
|
||||
|
||||
/** 直接问 worker(绕过 Manager)—— 证明实例真的落在**那台机器**上。 */
|
||||
const agent = async (base, path) => {
|
||||
const res = await fetch(base + path, { headers: { 'x-dsh-agent-token': TOKEN }, signal: AbortSignal.timeout(10_000) })
|
||||
const text = await res.text()
|
||||
return { status: res.status, body: text === '' ? null : JSON.parse(text) }
|
||||
}
|
||||
|
||||
/** 取页面:跟随重定向(dsh 首页 303),并对启动窗口的断连做重试。 */
|
||||
async function fetchPage(url, cookie, tries = 40) {
|
||||
let status = 0
|
||||
let snippet = ''
|
||||
for (let i = 0; i < tries; i += 1) {
|
||||
try {
|
||||
const res = await fetch(MANAGER + url, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(20_000) })
|
||||
status = res.status
|
||||
if (res.status === 200) {
|
||||
snippet = (await res.text()).slice(0, 200)
|
||||
break
|
||||
}
|
||||
} catch {
|
||||
status = 0
|
||||
}
|
||||
await sleep(1000)
|
||||
}
|
||||
return { status, snippet }
|
||||
}
|
||||
|
||||
async function waitRunning(cookie, tries = 60) {
|
||||
for (let i = 0; i < tries; i += 1) {
|
||||
const st = await json('/api/dsh/status', { cookie })
|
||||
if (st.body?.running === true) return st.body
|
||||
if (st.body?.instance?.status === 'crashed') return st.body
|
||||
await sleep(1000)
|
||||
}
|
||||
return await json('/api/dsh/status', { cookie }).then((r) => r.body)
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('=== 跨机演练:Manager=%s Worker=%s ===', MANAGER, AGENT1)
|
||||
|
||||
// ── 0) 两侧可达性(跨机链路的第一层证据)─────────────────────────────
|
||||
const h1 = await agent(AGENT1, '/healthz')
|
||||
assert(h1.status === 200 && h1.body.hostId === 'w-106', `Worker w-106 应可达(实际 ${JSON.stringify(h1.body)})`)
|
||||
assert(h1.body.tunnel?.ready === true, `Worker 侧隧道应就绪(实际 ${JSON.stringify(h1.body.tunnel)})`)
|
||||
console.log('⓪ worker 可达 -> %s(隧道 ready,已转发 %s)', h1.body.hostId, JSON.stringify(h1.body.tunnel.ports))
|
||||
|
||||
// ── 1) 管理面:注册 worker(join 脚本干的事)──────────────────────────
|
||||
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
|
||||
assert(adm.status === 200, `管理员登录失败 ${adm.status}`)
|
||||
const adminCookie = adm.setCookie.split(';')[0]
|
||||
|
||||
let r = await json('/api/admin/hosts', {
|
||||
method: 'POST',
|
||||
cookie: adminCookie,
|
||||
body: { id: 'w-106', endpoint: AGENT1, token: TOKEN, capacityMb: 4096 },
|
||||
})
|
||||
assert(r.status === 200, `注册 w-106 失败 ${r.status}`)
|
||||
const hosts = await json('/api/admin/hosts', { cookie: adminCookie })
|
||||
assert(hosts.body.hosts.some((h) => h.id === 'w-106'), 'w-106 出现在 worker 目录')
|
||||
assert(!('agentToken' in (hosts.body.hosts[0] ?? {})), '**绝不下发 agentToken**')
|
||||
console.log('① 注册 -> w-106(列表不含 agentToken)')
|
||||
|
||||
// ── 2) 用户流程 ───────────────────────────────────────────────────────
|
||||
const uname = `crossuser${Date.now() % 100000}`
|
||||
r = await json('/api/auth/register', { method: 'POST', body: { username: uname, password: USER_PW } })
|
||||
assert(r.status === 201, `注册应 201(实际 ${r.status})`)
|
||||
const users = await json('/api/admin/users', { cookie: adminCookie })
|
||||
const target = users.body.users.find((u) => u.username === uname)
|
||||
assert(target !== undefined, '管理员能看到待审用户')
|
||||
r = await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
|
||||
assert(r.status === 200, `审批应 200(实际 ${r.status})`)
|
||||
const login = await json('/api/auth/login', { method: 'POST', body: { username: uname, password: USER_PW } })
|
||||
assert(login.status === 200, `用户登录失败 ${login.status}`)
|
||||
const cookie = login.setCookie.split(';')[0]
|
||||
console.log('② 用户流程 -> 注册→审批→登录(uid=%s)', target.id)
|
||||
|
||||
// ── 3) 文件面跨机(Manager 在 47、目录落在 106)───────────────────────
|
||||
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
|
||||
assert(r.status === 200, `mkdir 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
console.log('③ 文件面 -> mkdir 经隧道落到 106 的 worker')
|
||||
|
||||
// ── 4) 拉起实例(真 dsh 在 **106** 上)────────────────────────────────
|
||||
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
|
||||
assert(r.status === 200, `launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
const st = await waitRunning(cookie)
|
||||
assert(st?.running === true, `实例应 running(实际 ${JSON.stringify(st)?.slice(0, 300)})`)
|
||||
const onAgent = await agent(AGENT1, '/instances')
|
||||
assert(onAgent.body.instances.length === 1, 'worker(106) 上有 1 个实例')
|
||||
console.log('④ 拉起 -> running=true,**实例在 106 上**(worker /instances=%d)', onAgent.body.instances.length)
|
||||
|
||||
// ── 5) 登录直达 + **跨机取页面**(本演练的核心证据)───────────────────
|
||||
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
|
||||
assert(enter.status === 200, `enter 应 200(实际 ${enter.status})`)
|
||||
const url = enter.body.url
|
||||
assert(typeof url === 'string' && url.includes('token='), `enter 应带 token(实际 ${url})`)
|
||||
const page = await fetchPage(url, cookie)
|
||||
assert(page.status === 200, `**跨机页面**应 200(实际 ${page.status})`)
|
||||
console.log('⑤ 跨机页面 -> 200(47 的 Manager 代理到 106 的实例;片段 %s)', page.snippet.replace(/\s+/g, ' ').slice(0, 60))
|
||||
|
||||
// ── 6) 第二台 worker(106 上模拟的第二台服务器)+ 迁移 ────────────────
|
||||
if (AGENT2 !== '') {
|
||||
const h2 = await agent(AGENT2, '/healthz')
|
||||
assert(h2.status === 200, `第二台 worker 应可达(实际 ${h2.status})`)
|
||||
r = await json('/api/admin/hosts', {
|
||||
method: 'POST',
|
||||
cookie: adminCookie,
|
||||
body: { id: 'w-106b', endpoint: AGENT2, token: TOKEN, capacityMb: 4096 },
|
||||
})
|
||||
assert(r.status === 200, `注册 w-106b 失败 ${r.status}`)
|
||||
console.log('⑥ 第二台 -> %s(模拟的第二台服务器)已注册', h2.body.hostId)
|
||||
|
||||
r = await json(`/api/admin/users/${target.id}/dsh/migrate`, {
|
||||
method: 'POST',
|
||||
cookie: adminCookie,
|
||||
body: { targetHost: 'w-106b' },
|
||||
})
|
||||
assert(r.status === 200, `迁移应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
assert(r.body.to === 'w-106b', `迁移目标应为 w-106b(实际 ${r.body.to})`)
|
||||
await waitRunning(cookie)
|
||||
const a1 = await agent(AGENT1, '/instances')
|
||||
const a2 = await agent(AGENT2, '/instances')
|
||||
assert(a1.body.instances.length === 0 && a2.body.instances.length === 1, '实例应从 w-106 移到 w-106b')
|
||||
console.log('⑦ 跨机迁移 -> %s → %s(epoch=%d),源机已空、目标机有 1 个实例', r.body.from, r.body.to, r.body.epoch)
|
||||
|
||||
const enter2 = await json('/api/dsh/enter', { method: 'POST', cookie })
|
||||
assert(enter2.status === 200 && enter2.body.url !== url, '迁移后 enter 应给**新** URL')
|
||||
const page2 = await fetchPage(enter2.body.url, cookie)
|
||||
assert(page2.status === 200, `迁移后页面应 200(实际 ${page2.status})`)
|
||||
console.log('⑧ 迁移后 -> 新 token URL 页面 200')
|
||||
} else {
|
||||
console.log('⑥⑦⑧ 跳过(未提供 AGENT2)')
|
||||
}
|
||||
|
||||
// ── 9) 收尾 ───────────────────────────────────────────────────────────
|
||||
r = await json('/api/dsh/stop', { method: 'POST', cookie })
|
||||
assert(r.status === 200, `stop 应 200(实际 ${r.status})`)
|
||||
console.log('⑨ 停止 -> ok')
|
||||
|
||||
console.log('\nOK: **真跨机**(47 当 Manager / 106 当 Worker,隧道跨界)演练通过')
|
||||
console.log(' ✓ worker 可达 ✓ 注册 ✓ 用户流程 ✓ 文件面跨机 ✓ 实例在 106 ✓ 跨机取页面 ✓ 跨 worker 迁移')
|
||||
} finally {
|
||||
/* 不主动清理:实例由调用方决定留或停(演练后要观察现场) */
|
||||
}
|
||||
@@ -0,0 +1,167 @@
|
||||
/**
|
||||
* T08 · **域名形态访问**验证(在演练环境做:不动生产、不动 DNS、不动证书)。
|
||||
*
|
||||
* 要回答的问题:生产切到 cluster(Manager 在 47、实例在 106)后,
|
||||
* **按域名形态访问**(`<用户名>.alotbuy.com`)还能不能正常落到 106 上的实例?
|
||||
*
|
||||
* 做法:给演练 Manager 设一个**测试 baseDomain**,用**显式 `Host` 头**打进去。
|
||||
*
|
||||
* ⚠️ 关键坑(2026-09-15 实际踩到,两次假阳性都源于它):**`fetch` 会静默丢弃 `Host` 头**
|
||||
* (Fetch 规范把它列为禁止头,undici 直接忽略)⇒ 请求落到"无租户"的门户路由、回 200 门户页,
|
||||
* 看起来"验证通过"其实是假的。⇒ **必须用 curl(`-H Host:`)**,且判据不能只看状态码。
|
||||
*
|
||||
* 运行(在 47 上):MANAGER=http://127.0.0.1:13080 BASE_DOMAIN=test.alotbuy.com \
|
||||
* AGENT=http://127.0.0.1:19000 AGENT_TOKEN=cross-machine-token \
|
||||
* node scripts/verify-cluster-domain.mjs
|
||||
*/
|
||||
import { execFileSync } from 'node:child_process'
|
||||
import { readFileSync } from 'node:fs'
|
||||
|
||||
const MANAGER = process.env.MANAGER ?? 'http://127.0.0.1:13080'
|
||||
const BASE_DOMAIN = process.env.BASE_DOMAIN ?? 'test.alotbuy.com'
|
||||
const AGENT = process.env.AGENT ?? 'http://127.0.0.1:19000'
|
||||
const TOKEN = process.env.AGENT_TOKEN ?? 'cross-machine-token'
|
||||
const ADMIN_PW = process.env.ADMIN_PW ?? 'crossmgr123'
|
||||
const USER_PW = 'domainuser123'
|
||||
|
||||
function assert(condition, message) {
|
||||
if (!condition) throw new Error('ASSERT: ' + message)
|
||||
}
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
||||
|
||||
const json = async (path, { method = 'GET', body, cookie } = {}) => {
|
||||
const res = await fetch(MANAGER + path, {
|
||||
method,
|
||||
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
})
|
||||
const text = await res.text()
|
||||
let parsed = null
|
||||
try {
|
||||
parsed = text === '' ? null : JSON.parse(text)
|
||||
} catch {
|
||||
parsed = { raw: text.slice(0, 200) }
|
||||
}
|
||||
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
|
||||
}
|
||||
|
||||
/**
|
||||
* 用 **curl** 带 `Host` 头取页面(`-L` 跟随重定向 ⇒ 等价真实浏览器)。
|
||||
* 返回 `{ status, body }`,body 从临时文件读(避免编码/二进制问题)。
|
||||
*/
|
||||
function getByHost(sub, path, cookie) {
|
||||
const out = '/tmp/dompage.html'
|
||||
const args = [
|
||||
'-s',
|
||||
'-L',
|
||||
'--max-time',
|
||||
'25',
|
||||
'-o',
|
||||
out,
|
||||
'-w',
|
||||
'%{http_code}',
|
||||
'-H',
|
||||
`Host: ${sub}.${BASE_DOMAIN}`,
|
||||
...(cookie ? ['-b', cookie] : []),
|
||||
`${MANAGER}${path}`,
|
||||
]
|
||||
let status = '0'
|
||||
try {
|
||||
status = execFileSync('curl', args, { encoding: 'utf8' }).trim()
|
||||
} catch {
|
||||
status = '0'
|
||||
}
|
||||
let body = ''
|
||||
try {
|
||||
body = readFileSync(out, 'utf8')
|
||||
} catch {
|
||||
body = ''
|
||||
}
|
||||
return { status: Number(status), body }
|
||||
}
|
||||
|
||||
/**
|
||||
* 判据:**dsh 实例页**带 `<base href="/">`(子路径与子域两种形态都带);平台门户页不带。
|
||||
* ⚠️ 只靠"含 dsh 字样"会把门户页误判成实例页(实测踩过这个假阳性)。
|
||||
*/
|
||||
const isDshApp = (html) => typeof html === 'string' && html.includes('<base href=')
|
||||
const describe = (html) => {
|
||||
const hit = []
|
||||
if (isDshApp(html)) hit.push('base-href')
|
||||
if (html.includes('/api/auth/login')) hit.push('platform-login')
|
||||
const t = /<title>([^<]*)<\/title>/.exec(html)
|
||||
return `${hit.join(',') || '(无特征)'} | title=${t === null ? '?' : t[1].trim()} | 首100字: ${html.replace(/\s+/g, ' ').slice(0, 100)}`
|
||||
}
|
||||
|
||||
async function waitRunning(cookie, tries = 60) {
|
||||
for (let i = 0; i < tries; i += 1) {
|
||||
const st = await json('/api/dsh/status', { cookie })
|
||||
if (st.body?.running === true) return true
|
||||
await sleep(1000)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('=== 域名形态验证:baseDomain=%s(Manager=%s)===', BASE_DOMAIN, MANAGER)
|
||||
|
||||
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
|
||||
assert(adm.status === 200, `管理员登录失败 ${adm.status}`)
|
||||
const adminCookie = adm.setCookie.split(';')[0]
|
||||
|
||||
const uname = `domuser${Date.now() % 100000}`
|
||||
let r = await json('/api/auth/register', { method: 'POST', body: { username: uname, password: USER_PW } })
|
||||
assert(r.status === 201, `注册应 201(实际 ${r.status})`)
|
||||
const users = await json('/api/admin/users', { cookie: adminCookie })
|
||||
const target = users.body.users.find((u) => u.username === uname)
|
||||
assert(target !== undefined, '能看到待审用户')
|
||||
await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
|
||||
const login = await json('/api/auth/login', { method: 'POST', body: { username: uname, password: USER_PW } })
|
||||
assert(login.status === 200, `用户登录失败 ${login.status}`)
|
||||
const cookie = login.setCookie.split(';')[0]
|
||||
console.log('① 用户 -> %s(uid=%s)', uname, target.id)
|
||||
|
||||
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
|
||||
assert(r.status === 200, `mkdir 失败 ${r.status}`)
|
||||
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
|
||||
assert(r.status === 200, `launch 失败 ${r.status} ${JSON.stringify(r.body)}`)
|
||||
assert(await waitRunning(cookie), '实例应 running')
|
||||
console.log('② 拉起 -> running=true')
|
||||
|
||||
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
|
||||
assert(enter.status === 200, `enter 失败 ${enter.status}`)
|
||||
const url = enter.body.url
|
||||
console.log('③ 直达 URL -> %s', url)
|
||||
assert(url.startsWith('https://'), `baseDomain 生效时应为 https://<子域>/(实际 ${url})`)
|
||||
assert(url.includes(`${uname}.${BASE_DOMAIN}`), `URL 应含用户名子域(实际 ${url})`)
|
||||
|
||||
// ④ 子域形态访问:**必须带 token**(真 dsh 没 token 只给自己的登录页)
|
||||
const token = new URL(url).searchParams.get('token') ?? ''
|
||||
assert(token !== '', `enter URL 应带 token(实际 ${url})`)
|
||||
let page = { status: 0, body: '' }
|
||||
for (let i = 0; i < 40; i += 1) {
|
||||
page = getByHost(uname, `/?token=${encodeURIComponent(token)}`, cookie)
|
||||
if (page.status === 200 && isDshApp(page.body)) break
|
||||
await sleep(1000)
|
||||
}
|
||||
assert(page.status === 200, `子域访问应 200(实际 ${page.status})`)
|
||||
assert(isDshApp(page.body), `子域访问必须是**真的 dsh 实例页**(实际 ${describe(page.body)})`)
|
||||
console.log('④ 子域访问 -> 200 且是**真 dsh 实例页**(Host: %s.%s → 106 上的实例)', uname, BASE_DOMAIN)
|
||||
|
||||
// ⑤ 越权对照:拿 A 的 cookie 访问**另一个真实用户**(root)的子域 ⇒ 必须 401/403
|
||||
const other = getByHost('root', `/?token=${encodeURIComponent(token)}`, cookie)
|
||||
assert(!isDshApp(other.body), `越权响应绝不能是实例页(实际 ${describe(other.body)})`)
|
||||
assert([401, 403].includes(other.status), `用 A 的 cookie 访问 root 子域应 401/403(实际 ${other.status})`)
|
||||
console.log('⑤ 越权对照 -> 用 A 的 cookie 访问 root 子域 = %d(正确拒绝)', other.status)
|
||||
|
||||
// ⑥ 独立取证:实例确实在 106
|
||||
const onAgent = await fetch(`${AGENT}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } }).then((x) => x.json())
|
||||
assert(onAgent.instances.length >= 1, 'worker(106) 上应有实例')
|
||||
console.log('⑥ 取证 -> 实例确实在 106(worker /instances=%d)', onAgent.instances.length)
|
||||
|
||||
await json('/api/dsh/stop', { method: 'POST', cookie })
|
||||
console.log('\nOK: **域名形态访问**在 cluster 下可用(子域 → Manager(47) → 实例(106)),且越权被拒')
|
||||
} finally {
|
||||
/* 现场保留 */
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
/**
|
||||
* T08 S5 · 跨机文件面验证。
|
||||
*
|
||||
* 关键设计:**Manager 的 `dataRoot` 故意与 worker 的 `dataRoot` 不同** ——
|
||||
* 只有这样"文件面真的走了远端"才被证明;若两个 root 相同,本地实现也能碰巧通过。
|
||||
*
|
||||
* 验的是:
|
||||
* ① 门户的路由(`/api/desktop/tree`、`/api/fs/*`)在 cluster 模式下照常工作(**路由零改动**);
|
||||
* ② 文件**落在 worker 的 dataRoot 下**、且**不在** Manager 的 dataRoot 下;
|
||||
* ③ 路径安全与本地**同源**(`bad_path` 走同一条 `resolveWithinRoot`);
|
||||
* ④ `resolvePath` 返回的是**实例眼里的路径**(按 worker 的 dataRoot 算)。
|
||||
*
|
||||
* 运行:node scripts/verify-cluster-fs.mjs
|
||||
*/
|
||||
import { existsSync, mkdtempSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { dirname, join } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
import { buildServer } from '../lib/web/server.js'
|
||||
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
|
||||
import { resolveConfig } from '../lib/config.js'
|
||||
import { hashPassword } from '../lib/web/auth.js'
|
||||
|
||||
function assert(condition, message) {
|
||||
if (!condition) throw new Error('ASSERT: ' + message)
|
||||
}
|
||||
|
||||
const here = dirname(fileURLToPath(import.meta.url))
|
||||
const fakeDsh = join(here, 'fake-dsh.mjs')
|
||||
const TOKEN = 'verify-cluster-fs-token'
|
||||
const workerRoot = mkdtempSync(join(tmpdir(), 'dsh-cfs-worker-'))
|
||||
const managerRoot = mkdtempSync(join(tmpdir(), 'dsh-cfs-manager-'))
|
||||
let agentApp
|
||||
let agentHandle
|
||||
let app
|
||||
|
||||
try {
|
||||
// ── worker agent(dataRoot = workerRoot)────────────────────────────────
|
||||
const agent = buildWorkerAgent(
|
||||
resolveConfig({ port: 0, dbPath: ':memory:', dataRoot: workerRoot, dshCommand: [process.execPath, fakeDsh], clusterHostId: 'w-1' }),
|
||||
{ hostId: 'w-1', token: TOKEN, port: 0, host: '127.0.0.1', instanceHost: '127.0.0.1', logLevel: 'warn' },
|
||||
)
|
||||
agentApp = agent.app
|
||||
agentHandle = agent
|
||||
await agentApp.listen({ host: '127.0.0.1', port: 0 })
|
||||
const agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
|
||||
console.log('worker -> dataRoot %s', workerRoot)
|
||||
|
||||
// ── Manager(dataRoot = managerRoot ≠ workerRoot;显式告知 worker 的 root)──
|
||||
app = await buildServer(
|
||||
resolveConfig({
|
||||
port: 0,
|
||||
dbPath: ':memory:',
|
||||
dataRoot: managerRoot,
|
||||
deployMode: 'cluster',
|
||||
clusterAgentUrl: agentUrl,
|
||||
clusterAgentToken: TOKEN,
|
||||
clusterInstanceHost: '127.0.0.1',
|
||||
clusterHostId: 'm-1',
|
||||
clusterWorkerDataRoot: workerRoot,
|
||||
}),
|
||||
)
|
||||
await app.listen({ port: 0 })
|
||||
const base = `http://127.0.0.1:${app.server.address().port}`
|
||||
console.log('manager -> dataRoot %s(与 worker 不同 ⇒ 能证明走远端)', managerRoot)
|
||||
|
||||
const json = async (path, { method = 'GET', body, cookie } = {}) => {
|
||||
const res = await fetch(base + path, {
|
||||
method,
|
||||
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
})
|
||||
const text = await res.text()
|
||||
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
|
||||
}
|
||||
|
||||
await app.db.createUser({
|
||||
id: 'u1',
|
||||
username: 'bob',
|
||||
passHash: await hashPassword('bobpass123'),
|
||||
role: 'active',
|
||||
homeDir: '/tmp/u1-home',
|
||||
})
|
||||
// 用户根必须建在 **worker 上**(这一步本身就走远端)
|
||||
await app.userFs.initUserRoot('u1')
|
||||
|
||||
// ④ resolvePath = 实例眼里的路径(按 worker 的 dataRoot)
|
||||
const resolved = app.userFs.resolvePath('u1', 'proj')
|
||||
assert(resolved === join(workerRoot, 'users', 'u1', 'ws', 'proj'), `resolvePath 应按 worker 的 root 计算(实际 ${resolved})`)
|
||||
console.log('④ resolvePath -> %s', resolved)
|
||||
|
||||
// ── ① 门户路由(零改动)──────────────────────────────────────────────
|
||||
let r = await json('/api/auth/login', { method: 'POST', body: { username: 'bob', password: 'bobpass123' } })
|
||||
assert(r.status === 200, 'login succeeds')
|
||||
const cookie = r.setCookie.split(';')[0]
|
||||
|
||||
r = await json('/api/desktop/tree', { cookie })
|
||||
assert(r.status === 200 && r.body.entries.length === 0, '空工作区列出 0 项')
|
||||
|
||||
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
|
||||
assert(r.status === 200, `mkdir 经远端成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
|
||||
r = await json('/api/fs/upload', {
|
||||
method: 'POST',
|
||||
cookie,
|
||||
body: { path: 'proj', name: 'hello.txt', data: Buffer.from('hi there').toString('base64') },
|
||||
})
|
||||
assert(r.status === 200, `upload 经远端成功(实际 ${r.status})`)
|
||||
|
||||
r = await json('/api/desktop/tree', { cookie })
|
||||
assert(r.status === 200 && r.body.entries.length === 1, '工作区里出现了 proj')
|
||||
console.log('① 门户路由 -> tree/mkdir/upload 全部经远端通过')
|
||||
|
||||
// ── ② 文件真的落在 worker 上 ──────────────────────────────────────────
|
||||
const onWorker = join(workerRoot, 'users', 'u1', 'ws', 'proj', 'hello.txt')
|
||||
const onManager = join(managerRoot, 'users', 'u1', 'ws', 'proj', 'hello.txt')
|
||||
assert(existsSync(onWorker), `文件应落在 worker:${onWorker}`)
|
||||
assert(!existsSync(onManager), `文件不该出现在 Manager 本地:${onManager}`)
|
||||
console.log('② 落点 -> worker 有、manager 无(确认走远端)')
|
||||
|
||||
// ── ③ 路径安全与本地同源 ──────────────────────────────────────────────
|
||||
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: '../evil' } })
|
||||
assert(r.status === 400 && r.body.error === 'bad_path', `越界路径应 400 bad_path(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
for (const bad of ['..', '../../etc']) {
|
||||
let threw = false
|
||||
try {
|
||||
app.userFs.resolvePath('u1', bad)
|
||||
} catch (err) {
|
||||
threw = err.code === 'bad_path'
|
||||
}
|
||||
assert(threw, `resolvePath(${bad}) 应抛 bad_path`)
|
||||
}
|
||||
console.log('③ 路径安全 -> bad_path 与本地同源(走同一个 resolveWithinRoot)')
|
||||
|
||||
// 下载回读(readFile 经远端)—— 注意该路由回的是**原始字节**,不是 JSON
|
||||
const dl = await fetch(`${base}/api/fs/download?path=proj/hello.txt`, { headers: { cookie } })
|
||||
assert(dl.status === 200, `download 经远端成功(实际 ${dl.status})`)
|
||||
const downloaded = await dl.text()
|
||||
assert(downloaded === 'hi there', `下载内容应为上传的原文(实际 ${JSON.stringify(downloaded)})`)
|
||||
console.log(' 下载回读 -> readFile 经远端成功(内容逐字一致)')
|
||||
|
||||
console.log('\nOK: 跨机文件面(RemoteUserFs → agent /fs/*)通过')
|
||||
console.log(' ✓ 门户路由零改动 ✓ 落在 worker ✓ 路径安全同源 ✓ resolvePath 按 worker 计算')
|
||||
} finally {
|
||||
await app?.close()
|
||||
await agentHandle?.stop()
|
||||
await new Promise((r) => setTimeout(r, 300))
|
||||
for (const dir of [workerRoot, managerRoot]) {
|
||||
try {
|
||||
rmSync(dir, { recursive: true, force: true })
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,208 @@
|
||||
/**
|
||||
* T08 S4 · 归属租约端到端验证(1a 形态:多 Manager + 一个 worker agent + 共享 PG)。
|
||||
*
|
||||
* 验的是 S4 的四条承重行为:
|
||||
* ① **归属真的落库**:launch 后 `dsh_instances` 有 `host_id` / `epoch` / `lease_until`;
|
||||
* ② **单写者**:另一个 Manager(另一个 worker 身份)在租约存活期内**拉不起来**同一用户
|
||||
* ⇒ 抛 `LeaseBusyError`(退让,不是接管);
|
||||
* ③ **stop 释放归属** ⇒ 别人立刻能接(不用等 TTL);
|
||||
* ④ **失权即 self-fence**:租约被抢走后,原持有者下一次心跳会把"更高 epoch"下发给 worker,
|
||||
* 由 worker **停掉自己那个实例**(防双写的最后一道防线)。
|
||||
* ⑤ 顺带验**注册 + 心跳**:`dsh_hosts` 里有记录且 `last_heartbeat` 持续更新。
|
||||
*
|
||||
* 需要 PG(两个 Manager 必须共享 DB,否则谈不上"归属"):
|
||||
* CLUSTER_TEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node scripts/verify-cluster-lease.mjs
|
||||
*/
|
||||
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { dirname, join } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
import pg from 'pg'
|
||||
import { buildServer } from '../lib/web/server.js'
|
||||
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
|
||||
import { resolveConfig } from '../lib/config.js'
|
||||
import { hashPassword } from '../lib/web/auth.js'
|
||||
|
||||
function assert(condition, message) {
|
||||
if (!condition) throw new Error('ASSERT: ' + message)
|
||||
}
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
||||
|
||||
const PG_URL = process.env.CLUSTER_TEST_PG_URL
|
||||
if (PG_URL === undefined || PG_URL === '') {
|
||||
console.error('需要 CLUSTER_TEST_PG_URL(两个 Manager 必须共享同一个库)')
|
||||
process.exit(2)
|
||||
}
|
||||
const TOKEN = 'verify-cluster-lease-token'
|
||||
const here = dirname(fileURLToPath(import.meta.url))
|
||||
const fakeDsh = join(here, 'fake-dsh.mjs')
|
||||
const dataRoot = mkdtempSync(join(tmpdir(), 'dsh-lease-'))
|
||||
|
||||
// 短 TTL:TTL=1200ms > 2×renew=500ms(满足不变量),便于在秒级制造"过期/被抢"
|
||||
process.env.DSHS_CLUSTER_LEASE_TTL_MS = '1200'
|
||||
process.env.DSHS_CLUSTER_LEASE_RENEW_MS = '500'
|
||||
|
||||
let agentApp
|
||||
let agentHandle
|
||||
const managers = []
|
||||
|
||||
/** 清空测试库里的三张表(该库专供本测试)。 */
|
||||
async function resetPg() {
|
||||
const client = new pg.Client({ connectionString: PG_URL })
|
||||
await client.connect()
|
||||
await client.query('DELETE FROM dsh_instances')
|
||||
await client.query('DELETE FROM dsh_hosts')
|
||||
await client.query('DELETE FROM users')
|
||||
await client.end()
|
||||
}
|
||||
|
||||
/** 起一个 Manager(cluster 模式)。hostId 即"它绑定的 worker 身份"。 */
|
||||
async function startManager(hostId) {
|
||||
const app = await buildServer(
|
||||
resolveConfig({
|
||||
port: 0,
|
||||
dbUrl: PG_URL,
|
||||
dataRoot,
|
||||
deployMode: 'cluster',
|
||||
clusterAgentUrl: agentUrl,
|
||||
clusterAgentToken: TOKEN,
|
||||
clusterInstanceHost: '127.0.0.1',
|
||||
clusterHostId: hostId,
|
||||
}),
|
||||
)
|
||||
await app.listen({ port: 0 })
|
||||
managers.push(app)
|
||||
return app
|
||||
}
|
||||
|
||||
let agentUrl = ''
|
||||
|
||||
try {
|
||||
await resetPg()
|
||||
|
||||
// ── 0) worker agent ───────────────────────────────────────────────────
|
||||
const agentConfig = resolveConfig({
|
||||
port: 0,
|
||||
dbPath: ':memory:',
|
||||
dataRoot,
|
||||
dshCommand: [process.execPath, fakeDsh],
|
||||
clusterHostId: 'w-1',
|
||||
})
|
||||
const agent = buildWorkerAgent(agentConfig, {
|
||||
hostId: 'w-1',
|
||||
token: TOKEN,
|
||||
port: 0,
|
||||
host: '127.0.0.1',
|
||||
instanceHost: '127.0.0.1',
|
||||
logLevel: 'warn',
|
||||
})
|
||||
agentApp = agent.app
|
||||
agentHandle = agent
|
||||
await agentApp.listen({ host: '127.0.0.1', port: 0 })
|
||||
agentUrl = `http://127.0.0.1:${agentApp.server.address().port}`
|
||||
const agentJson = async (path, { method = 'GET', body } = {}) => {
|
||||
const res = await fetch(agentUrl + path, {
|
||||
method,
|
||||
headers: { [AGENT_TOKEN_HEADER]: TOKEN, ...(body ? { 'content-type': 'application/json' } : {}) },
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
})
|
||||
const text = await res.text()
|
||||
return { status: res.status, body: text ? JSON.parse(text) : null }
|
||||
}
|
||||
|
||||
// ── 1) 两个 Manager(不同 worker 身份)────────────────────────────────
|
||||
const m1 = await startManager('m-1')
|
||||
const m2 = await startManager('m-2')
|
||||
console.log('manager -> m-1 %s / m-2 %s(共享 PG + 同一 agent)', m1.supervisor.hostId, m2.supervisor.hostId)
|
||||
|
||||
await m1.db.createUser({
|
||||
id: 'u1',
|
||||
username: 'carol',
|
||||
passHash: await hashPassword('carolpass123'),
|
||||
role: 'active',
|
||||
homeDir: '/tmp/u1-home',
|
||||
})
|
||||
mkdirSync(join(dataRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
|
||||
const folder = join(dataRoot, 'users', 'u1', 'ws', 'proj')
|
||||
|
||||
// ⑤ 注册 + 心跳:dsh_hosts 里应有记录(启动即注册 + 立即一次心跳)
|
||||
const hosts = await m1.db.listDshHosts()
|
||||
assert(hosts.length === 2, `dsh_hosts 应有 2 条(实际 ${hosts.length})`)
|
||||
assert(hosts.every((h) => h.status === 'up' && h.lastHeartbeat !== null), 'worker 状态 up 且有心跳时间')
|
||||
console.log('注册/心跳 -> dsh_hosts =', hosts.map((h) => `${h.id}:${h.status}`).join(', '))
|
||||
|
||||
// ── 2) m-1 拉起:归属必须落库 ─────────────────────────────────────────
|
||||
const inst1 = await m1.supervisor.launch('u1', folder)
|
||||
assert(inst1.userId === 'u1', 'launch 返回实例')
|
||||
let row = await m1.db.findUserInstance('u1', 'main')
|
||||
assert(row.hostId === 'm-1', `归属应落库为 m-1(实际 ${row.hostId})`)
|
||||
assert(row.epoch === 1, `首次抢占 epoch 应为 1(实际 ${row.epoch})`)
|
||||
assert(row.leaseUntil > Date.now(), 'lease_until 应在未来')
|
||||
console.log('① 归属落库 -> host_id=%s epoch=%d lease_until=+%dms', row.hostId, row.epoch, row.leaseUntil - Date.now())
|
||||
|
||||
// worker 侧真的收到了 epoch=1(`/fence` 同值 ⇒ 不该被 fence)
|
||||
const f0 = await agentJson('/fence', { method: 'POST', body: { userId: 'u1', epoch: 1 } })
|
||||
assert(f0.body.fenced === false, 'worker 已记录 epoch=1(同值不 fence)')
|
||||
console.log(' worker 已记录 epoch=1')
|
||||
|
||||
// ── 3) 单写者:m-2 在租约存活期内拉不起来 ────────────────────────────
|
||||
let busy
|
||||
try {
|
||||
await m2.supervisor.launch('u1', folder)
|
||||
} catch (err) {
|
||||
busy = err
|
||||
}
|
||||
assert(busy !== undefined, 'm-2 必须拉起失败')
|
||||
assert(busy.name === 'LeaseBusyError', `应是 LeaseBusyError(实际 ${busy.name})`)
|
||||
assert(busy.holder === 'm-1', `错误里应带持有者 m-1(实际 ${busy.holder})`)
|
||||
const stillMine = await m1.db.findUserInstance('u1', 'main')
|
||||
assert(stillMine.hostId === 'm-1' && stillMine.epoch === 1, '失败方不得改动归属')
|
||||
console.log('② 单写者 -> m-2 抛 LeaseBusyError(holder=%s),归属未被改动', busy.holder)
|
||||
|
||||
// ── 4) stop 释放 ⇒ 别人立刻能接(不用等 TTL)──────────────────────────
|
||||
await m1.supervisor.stop('u1')
|
||||
row = await m1.db.findUserInstance('u1', 'main')
|
||||
assert(row.hostId === null, 'stop 后归属应清空')
|
||||
|
||||
// 多机语义(S6 起):不显式指定目标机时由 `selectHost` 按容量挑"最优的那台",
|
||||
// 不一定是 m-2 ⇒ 这一步要验的是"释放后可被接管",所以**显式指定 m-2**(确定性)。
|
||||
const inst2 = await m2.supervisor.launch('u1', folder, undefined, { hostId: 'm-2' })
|
||||
assert(inst2.userId === 'u1', 'm-2 拉起成功')
|
||||
row = await m2.db.findUserInstance('u1', 'main')
|
||||
assert(row.hostId === 'm-2', `归属应转给 m-2(实际 ${row.hostId})`)
|
||||
assert(row.epoch === 2, `epoch 必须递增到 2(实际 ${row.epoch})`)
|
||||
console.log('③ 释放即接手 -> host_id=m-2 epoch=%d(epoch 单调递增)', row.epoch)
|
||||
|
||||
// ── 5) 失权即 self-fence ─────────────────────────────────────────────
|
||||
// 让 m-2 也"死掉"(停心跳)→ 等待 TTL 过期 → m-1 抢占(epoch=3)
|
||||
m2.supervisor.stopHeartbeat()
|
||||
await sleep(1400)
|
||||
const stolen = await m1.db.claimInstance('u1', 'm-1', 60_000)
|
||||
assert(stolen.ok === true && stolen.epoch === 3, `m-1 过期后应能抢到 epoch=3(实际 ${JSON.stringify(stolen)})`)
|
||||
console.log(' m-1 在租约过期后抢回(epoch=3)')
|
||||
|
||||
// m-2 的下一跳心跳发现自己失权 ⇒ 给 worker 下发更高 epoch ⇒ worker 停掉自己那个实例
|
||||
const before = await agentJson('/instances')
|
||||
await m2.supervisor.tick()
|
||||
await sleep(200)
|
||||
const after = await agentJson('/instances')
|
||||
assert(after.body.instances.length === 0, `失权方心跳后实例应被停(前 ${before.body.instances.length} → 后 ${after.body.instances.length})`)
|
||||
console.log('④ 失权即 fence -> worker 实例数 %d → %d(self-fencing 生效)', before.body.instances.length, after.body.instances.length)
|
||||
|
||||
// 归属仍在 m-1 名下(fence 不会误清他人的归属)
|
||||
row = await m1.db.findUserInstance('u1', 'main')
|
||||
assert(row.hostId === 'm-1', 'fence 不该清掉持有者的归属')
|
||||
console.log(' 归属仍在 m-1 名下(未被误清)')
|
||||
|
||||
console.log('\nOK: 归属租约(1a 形态)端到端通过')
|
||||
console.log(' ✓ 归属落库 ✓ 单写者(LeaseBusyError) ✓ 释放即接手 ✓ 失权即 self-fence ✓ 注册+心跳')
|
||||
} finally {
|
||||
for (const app of managers) await app?.close()
|
||||
await agentHandle?.stop()
|
||||
await sleep(500)
|
||||
try {
|
||||
rmSync(dataRoot, { recursive: true, force: true })
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,365 @@
|
||||
/**
|
||||
* T08 · **真实部署**端到端功能确认(不是单进程内测试)。
|
||||
*
|
||||
* 与 `verify-cluster-*.mjs` 的区别(那些是**组件级**验证,两个 Fastify 跑在同一进程里):
|
||||
* 这里**真的起进程** —— 2 个 `dshs worker` agent + 1 个 Manager 都是独立进程,
|
||||
* 经**真 HTTP**(127.0.0.1 端口)与**真 PG** 通信,实例是**真 `dsh` 子进程**(account 隔离)。
|
||||
* 它回答的是最后一个问题:**这套东西按部署形态装起来,到底能不能用。**
|
||||
*
|
||||
* 检查链路(一条真实的用户路径):
|
||||
* ① bootstrap-admin → ③ register → approve(平台现有流程)
|
||||
* ④ 登录 → ⑤ 建文件夹(经 RemoteUserFs 落到 worker)→ ⑥ launch(经 agent 起真 dsh)
|
||||
* ⑦ 轮询 status → ⑧ **取实例页面(经代理)**← 功能确认的关键一步
|
||||
* ⑨ stop → ⑩ 起第二台 worker → 注册 → **迁移** → 再取一次页面
|
||||
* ⑪ `dshs doctor` / `dshs cluster status`(观测面)
|
||||
*
|
||||
* 需要:106 上 PG 已在 127.0.0.1:15432;以 root 运行(account 隔离要 setpriv/systemd-run)。
|
||||
* 运行:CLUSTER_LIVE_PG_URL=postgres://dshs:[email protected]:15432/dshs_live node scripts/verify-cluster-live.mjs
|
||||
*/
|
||||
import { spawn, spawnSync } from 'node:child_process'
|
||||
import { createWriteStream, mkdirSync, readFileSync, rmSync } from 'node:fs'
|
||||
import { dirname, join } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
|
||||
function assert(condition, message) {
|
||||
if (!condition) throw new Error('ASSERT: ' + message)
|
||||
}
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
||||
|
||||
const PG_URL = process.env.CLUSTER_LIVE_PG_URL
|
||||
if (PG_URL === undefined || PG_URL === '') {
|
||||
console.error('需要 CLUSTER_LIVE_PG_URL')
|
||||
process.exit(2)
|
||||
}
|
||||
const here = dirname(fileURLToPath(import.meta.url))
|
||||
const repoRoot = join(here, '..')
|
||||
const CLI = join(repoRoot, 'lib', 'cli.js')
|
||||
const TOKEN = 'live-cluster-agent-token'
|
||||
const DATA_ROOT = process.env.CLUSTER_LIVE_DATA_ROOT ?? '/opt/dshs-cluster/live-data'
|
||||
const ISO = process.env.CLUSTER_LIVE_ISOLATION ?? 'account'
|
||||
const M_PORT = Number(process.env.CLUSTER_LIVE_MANAGER_PORT ?? 13080)
|
||||
const A1_PORT = Number(process.env.CLUSTER_LIVE_AGENT1_PORT ?? 19000)
|
||||
const A2_PORT = Number(process.env.CLUSTER_LIVE_AGENT2_PORT ?? 19001)
|
||||
const ADMIN_PW = 'liveadmin123'
|
||||
const USER_PW = 'liveuser123'
|
||||
|
||||
const procs = []
|
||||
/** 起一个子进程并记下来(收尾统一 SIGTERM ⇒ agent 会先 teardown 实例再退出)。 */
|
||||
function run(label, args, env) {
|
||||
const child = spawn(process.execPath, args, {
|
||||
cwd: repoRoot,
|
||||
env: { ...process.env, ...env },
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
})
|
||||
// ⚠️ **必须留日志**:不留就只能看到"实例没了"而看不到为什么(2026-09-15 实测踩到)
|
||||
const logPath = `/tmp/live-${label}.log`
|
||||
const stream = createWriteStream(logPath, { flags: 'w' })
|
||||
child.stdout.pipe(stream)
|
||||
child.stderr.pipe(stream)
|
||||
child.on('exit', (code) => {
|
||||
if (code !== null && code !== 0 && !stopping) console.error(`[${label}] 提前退出 code=${code}(日志 ${logPath})`)
|
||||
})
|
||||
procs.push({ label, child, logPath })
|
||||
return child
|
||||
}
|
||||
|
||||
/** 打印某个子进程日志的尾部(诊断用)。 */
|
||||
function tailLog(label, lines = 12) {
|
||||
const found = procs.find((p) => p.label === label)
|
||||
if (found === undefined) return
|
||||
try {
|
||||
const text = readFileSync(found.logPath, 'utf8').trimEnd().split('\n')
|
||||
console.error(` ── ${label} 日志尾部 ──`)
|
||||
for (const line of text.slice(-lines)) console.error(' ', line.slice(0, 200))
|
||||
} catch {
|
||||
/* 没日志就算了 */
|
||||
}
|
||||
}
|
||||
|
||||
let stopping = false
|
||||
function shutdownAll() {
|
||||
stopping = true
|
||||
for (const { child } of procs) {
|
||||
try {
|
||||
child.kill('SIGTERM')
|
||||
} catch {
|
||||
/* 已退出 */
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** 轮询等一个 URL 可用。 */
|
||||
async function waitHttp(url, timeoutMs, what) {
|
||||
const deadline = Date.now() + timeoutMs
|
||||
let last = ''
|
||||
while (Date.now() < deadline) {
|
||||
try {
|
||||
const res = await fetch(url, { signal: AbortSignal.timeout(2000) })
|
||||
if (res.status < 500) return res.status
|
||||
last = `HTTP ${res.status}`
|
||||
} catch (err) {
|
||||
last = err instanceof Error ? err.message : String(err)
|
||||
}
|
||||
await sleep(300)
|
||||
}
|
||||
throw new Error(`等待 ${what} 超时(${url}):${last}`)
|
||||
}
|
||||
|
||||
const base = `http://127.0.0.1:${M_PORT}`
|
||||
const json = async (path, { method = 'GET', body, cookie } = {}) => {
|
||||
const res = await fetch(base + path, {
|
||||
method,
|
||||
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
})
|
||||
const text = await res.text()
|
||||
let parsed = null
|
||||
try {
|
||||
parsed = text === '' ? null : JSON.parse(text)
|
||||
} catch {
|
||||
parsed = { raw: text.slice(0, 200) }
|
||||
}
|
||||
return { status: res.status, body: parsed, setCookie: res.headers.get('set-cookie') }
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('=== 真实部署检查:dataRoot=%s 隔离=%s ===', DATA_ROOT, ISO)
|
||||
rmSync(DATA_ROOT, { recursive: true, force: true })
|
||||
mkdirSync(DATA_ROOT, { recursive: true })
|
||||
|
||||
// 干净的 PG 库(本检查专用)
|
||||
const psql = (sql) =>
|
||||
spawnSync('su', ['-', 'postgres', '-c', `/usr/bin/psql -p 15432 -q -c "${sql}"`], { encoding: 'utf8' })
|
||||
psql('DROP DATABASE IF EXISTS dshs_live')
|
||||
psql('CREATE DATABASE dshs_live OWNER dshs')
|
||||
console.log('PG -> dshs_live 已重建')
|
||||
|
||||
// ── ① worker agent(独立进程)─────────────────────────────────────────
|
||||
const agentEnv = { DSHS_DATA_ROOT: DATA_ROOT, DSHS_ISOLATION_MODE: ISO }
|
||||
run('agent-w-1', [CLI, 'worker', '--token', TOKEN, '--port', String(A1_PORT), '--host', '127.0.0.1', '--host-id', 'w-1', '--instance-host', '127.0.0.1', '--log-level', 'warn'], agentEnv)
|
||||
await waitHttp(`http://127.0.0.1:${A1_PORT}/healthz`, 15_000, 'agent w-1')
|
||||
console.log('worker w-1 -> http://127.0.0.1:%d 就绪', A1_PORT)
|
||||
|
||||
// ── ② Manager(独立进程,cluster 模式)────────────────────────────────
|
||||
const managerEnv = {
|
||||
DSHS_DEPLOY_MODE: 'cluster',
|
||||
DSHS_DB_URL: PG_URL,
|
||||
DSHS_DATA_ROOT: DATA_ROOT,
|
||||
DSHS_CLUSTER_HOST_ID: 'm-1',
|
||||
DSHS_CLUSTER_AGENT_URL: `http://127.0.0.1:${A1_PORT}`,
|
||||
DSHS_CLUSTER_AGENT_TOKEN: TOKEN,
|
||||
DSHS_CLUSTER_INSTANCE_HOST: '127.0.0.1',
|
||||
DSHS_CLUSTER_WORKER_DATA_ROOT: DATA_ROOT,
|
||||
DSHS_CLUSTER_CAPACITY_MB: '-1', // Manager 自己**不承载实例**
|
||||
DSHS_CLUSTER_REGISTER_SELF: '0', // 专用 Manager ⇒ **不自注册**(一个 agent 只应有一条 host 记录)
|
||||
DSHS_CLUSTER_LEASE_TTL_MS: '30000',
|
||||
}
|
||||
run('manager', [CLI, '--port', String(M_PORT), '--host', '127.0.0.1', '--log-level', 'warn'], managerEnv)
|
||||
await waitHttp(`${base}/login.html`, 20_000, 'Manager')
|
||||
console.log('manager -> %s 就绪(deployMode=cluster)', base)
|
||||
|
||||
// ── ③ bootstrap-admin(首次建管理员)──────────────────────────────────
|
||||
const boot = spawnSync(process.execPath, [CLI, 'bootstrap-admin', '--username', 'root', '--password', ADMIN_PW], {
|
||||
cwd: repoRoot,
|
||||
// 用 local 模式初始化管理员 root:它就是这台机上的目录,与 worker 用同一个 DATA_ROOT
|
||||
env: { ...process.env, DSHS_DATA_ROOT: DATA_ROOT, DSHS_DB_URL: PG_URL },
|
||||
encoding: 'utf8',
|
||||
})
|
||||
assert(boot.status === 0, `bootstrap-admin 失败:${boot.stderr?.slice(0, 300)}`)
|
||||
console.log('管理员 -> root 已创建(bootstrap-admin)')
|
||||
|
||||
// ── ④ 真实用户流程:注册 → 审批 → 登录 ────────────────────────────────
|
||||
let adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: ADMIN_PW } })
|
||||
assert(adm.status === 200, `管理员登录失败:${adm.status}`)
|
||||
const adminCookie = adm.setCookie.split(';')[0]
|
||||
|
||||
let r = await json('/api/auth/register', { method: 'POST', body: { username: 'liveuser', password: USER_PW } })
|
||||
assert(r.status === 201, `注册应 201(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
const users = await json('/api/admin/users', { cookie: adminCookie })
|
||||
const target = users.body.users.find((u) => u.username === 'liveuser')
|
||||
assert(target !== undefined, '管理员能列出待审用户')
|
||||
r = await json(`/api/admin/users/${target.id}/approve`, { method: 'POST', cookie: adminCookie })
|
||||
assert(r.status === 200, `审批应 200(实际 ${r.status})`)
|
||||
|
||||
const login = await json('/api/auth/login', { method: 'POST', body: { username: 'liveuser', password: USER_PW } })
|
||||
assert(login.status === 200, `用户登录失败:${login.status}`)
|
||||
const cookie = login.setCookie.split(';')[0]
|
||||
console.log('① 用户流程 -> 注册 → 审批 → 登录 全部通过(uid=%s)', target.id)
|
||||
|
||||
// 显式注册 w-1(= join 脚本那一步:agent 已在跑,调管理面登记)
|
||||
r = await json('/api/admin/hosts', {
|
||||
method: 'POST',
|
||||
cookie: adminCookie,
|
||||
body: { id: 'w-1', endpoint: `http://127.0.0.1:${A1_PORT}`, token: TOKEN, capacityMb: 4096 },
|
||||
})
|
||||
assert(r.status === 200, `注册 w-1 应 200(实际 ${r.status})`)
|
||||
|
||||
// ── ⑤ 建文件夹(经 RemoteUserFs 落到 worker)─────────────────────────
|
||||
r = await json('/api/fs/mkdir', { method: 'POST', cookie, body: { path: 'proj' } })
|
||||
assert(r.status === 200, `mkdir 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
assert(
|
||||
(await json('/api/desktop/tree', { cookie })).body.entries.some((e) => e.name === 'proj'),
|
||||
'工作区里出现 proj',
|
||||
)
|
||||
console.log('② 文件面 -> mkdir 落库到 worker(经 agent /fs/mkdir)')
|
||||
|
||||
// ── ⑥ 拉起实例(经 agent 起**真 dsh**)───────────────────────────────
|
||||
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
|
||||
assert(r.status === 200, `launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
assert(typeof r.body.url === 'string' && r.body.url.startsWith('/u/'), `launch 返回子路径形态 URL(实际 ${r.body.url})`)
|
||||
console.log('③ 拉起 -> %s(真 dsh 启动中,token 稍后才吐)', r.body.url)
|
||||
// 直接问 worker(绕过 Manager):实例到底在不在 agent 手上
|
||||
const onAgent = await fetch(`http://127.0.0.1:${A1_PORT}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } })
|
||||
console.log(' worker 视角 -> /instances = %s', (await onAgent.text()).slice(0, 200))
|
||||
|
||||
// ── ⑦ 轮询到「running **且** launch token 到位」 ─────────────────────
|
||||
// 真 dsh 与 fake-dsh 不同:进程起来 ≠ 已打印 token。**launch token 回传(P0-6)**
|
||||
// 只有在 token 真的从 worker 传回 Manager 之后才算成立,所以这里要等到它。
|
||||
const t0 = Date.now()
|
||||
let running = false
|
||||
let tokenSeen = ''
|
||||
let lastStatus = null
|
||||
for (let i = 0; i < 90; i += 1) {
|
||||
const st = await json('/api/dsh/status', { cookie })
|
||||
lastStatus = st.body
|
||||
const main = st.body.main
|
||||
if (main?.status === 'crashed') {
|
||||
console.error(' 实例崩溃:exitCode=%s lastError=%s', main.exitCode, String(main.lastError).slice(0, 400))
|
||||
break
|
||||
}
|
||||
// 注意:`/api/dsh/status` 的实例视图是**精简视图**(id/port/status/restarts),
|
||||
// **不含 launchToken** ⇒ token 的存在性用下面的 `/api/dsh/enter` 判定(它回带 token 的 URL)。
|
||||
if (st.body.running === true) {
|
||||
running = true
|
||||
tokenSeen = st.body.url ?? ''
|
||||
break
|
||||
}
|
||||
await sleep(1000)
|
||||
}
|
||||
if (!running) {
|
||||
tailLog('agent-w-1', 20)
|
||||
tailLog('manager', 10)
|
||||
}
|
||||
assert(running, `实例应在 90s 内 running(实际 ${JSON.stringify(lastStatus)?.slice(0, 500)})`)
|
||||
console.log('④ 状态 -> running=true(真 dsh,隔离=%s,耗时 %ds)', ISO, Math.round((Date.now() - t0) / 1000))
|
||||
|
||||
// ── ⑧ **登录直达**(P0-6 的真实端到端):enter 走"复用已运行实例"分支 ⇒ 带 token 的 URL
|
||||
const enter = await json('/api/dsh/enter', { method: 'POST', cookie })
|
||||
assert(enter.status === 200, `enter 应 200(实际 ${enter.status} ${JSON.stringify(enter.body)})`)
|
||||
const launchUrl = enter.body.url
|
||||
assert(typeof launchUrl === 'string' && launchUrl.includes('token='), `enter 应返回**带 token** 的直达 URL(实际 ${launchUrl})`)
|
||||
console.log('⑤ 登录直达 -> %s', launchUrl)
|
||||
|
||||
let pageStatus = 0
|
||||
let pageSnippet = ''
|
||||
for (let i = 0; i < 40; i += 1) {
|
||||
// 真实浏览器会**跟随重定向**(dsh 首页 303 → 应用页)⇒ 这里也跟随,否则会误判为失败。
|
||||
// ⚠️ 必须 try/catch:实例刚 spawn 时正在初始化,代理可能中途断连(`other side closed`),
|
||||
// 这是**启动窗口的正常现象**,重试即可 —— 不捕获会让检查在第一次尝试就失败。
|
||||
try {
|
||||
const res = await fetch(base + launchUrl, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(15_000) })
|
||||
pageStatus = res.status
|
||||
if (res.status === 200) {
|
||||
pageSnippet = (await res.text()).slice(0, 400)
|
||||
break
|
||||
}
|
||||
} catch {
|
||||
pageStatus = 0
|
||||
}
|
||||
await sleep(1000)
|
||||
}
|
||||
assert(pageStatus === 200, `实例页面应最终 200(实际 ${pageStatus})`)
|
||||
console.log('⑥ 实例页面 -> 200(经 Manager 代理到 worker 上 account 沙箱内的真 dsh;已跟随 303 重定向)')
|
||||
|
||||
// ── ⑨ 停止 ────────────────────────────────────────────────────────────
|
||||
r = await json('/api/dsh/stop', { method: 'POST', cookie })
|
||||
assert(r.status === 200, `stop 应 200(实际 ${r.status})`)
|
||||
await sleep(500)
|
||||
assert((await json('/api/dsh/status', { cookie })).body.running === false, 'stop 后 running=false')
|
||||
console.log('⑦ 停止 -> ok')
|
||||
|
||||
// ── ⑩ 第二台 worker + 迁移 ────────────────────────────────────────────
|
||||
run('agent-w-2', [CLI, 'worker', '--token', TOKEN, '--port', String(A2_PORT), '--host', '127.0.0.1', '--host-id', 'w-2', '--instance-host', '127.0.0.1', '--log-level', 'warn'], agentEnv)
|
||||
await waitHttp(`http://127.0.0.1:${A2_PORT}/healthz`, 15_000, 'agent w-2')
|
||||
r = await json('/api/admin/hosts', {
|
||||
method: 'POST',
|
||||
cookie: adminCookie,
|
||||
body: { id: 'w-2', endpoint: `http://127.0.0.1:${A2_PORT}`, token: TOKEN, capacityMb: 4096 },
|
||||
})
|
||||
assert(r.status === 200, `注册 w-2 应 200(实际 ${r.status})`)
|
||||
const hosts = await json('/api/admin/hosts', { cookie: adminCookie })
|
||||
assert(hosts.body.hosts.some((h) => h.id === 'w-2'), 'w-2 出现在 worker 目录')
|
||||
assert(!('agentToken' in (hosts.body.hosts[0] ?? {})), '**绝不下发 agentToken**')
|
||||
console.log('⑧ 第二台 -> w-2 已注册(且列表不含 agentToken)')
|
||||
|
||||
// 重新拉起(此刻只有 w-1 是候选 ⇒ 确定性落 w-1),再注册 w-2、再迁移
|
||||
r = await json('/api/dsh/launch', { method: 'POST', cookie, body: { folder: 'proj' } })
|
||||
assert(r.status === 200, `再次 launch 应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
for (let i = 0; i < 40 && !(await json('/api/dsh/status', { cookie })).body.running; i += 1) await sleep(1000)
|
||||
const owner = (await json('/api/dsh/status', { cookie })).body
|
||||
assert(owner.running === true, '重新拉起后 running=true')
|
||||
|
||||
r = await json(`/api/admin/users/${target.id}/dsh/migrate`, {
|
||||
method: 'POST',
|
||||
cookie: adminCookie,
|
||||
body: { targetHost: 'w-2' },
|
||||
})
|
||||
assert(r.status === 200, `迁移应 200(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
assert(r.body.to === 'w-2', `迁移目标应为 w-2(实际 ${r.body.to})`)
|
||||
for (let i = 0; i < 30 && !(await json('/api/dsh/status', { cookie })).body.running; i += 1) await sleep(1000)
|
||||
console.log('⑨ 迁移 -> %s → %s(epoch=%d)', r.body.from, r.body.to, r.body.epoch)
|
||||
|
||||
// ⚠️ 迁移后实例是**新进程 ⇒ 新 launch token**:旧 URL 里的 token 已失效(404 是**预期**行为)。
|
||||
// 真实用户会重新走 `/api/dsh/enter`(门户的"进入工作区"就是这个接口)拿**新** URL ⇒ 这里照做。
|
||||
let newUrl = ''
|
||||
pageStatus = 0
|
||||
for (let i = 0; i < 40; i += 1) {
|
||||
try {
|
||||
const en = await json('/api/dsh/enter', { method: 'POST', cookie })
|
||||
if (en.status === 200 && typeof en.body.url === 'string') {
|
||||
newUrl = en.body.url
|
||||
const res = await fetch(base + newUrl, { headers: { cookie }, redirect: 'follow', signal: AbortSignal.timeout(15_000) })
|
||||
pageStatus = res.status
|
||||
if (res.status === 200) {
|
||||
pageSnippet = (await res.text()).slice(0, 200)
|
||||
break
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
pageStatus = 0
|
||||
}
|
||||
await sleep(1000)
|
||||
}
|
||||
if (newUrl === '' || newUrl === launchUrl) {
|
||||
// 诊断:两台 agent 各自认为有什么 + DB 归属如何
|
||||
for (const [label, port] of [['w-1', A1_PORT], ['w-2', A2_PORT]]) {
|
||||
const res = await fetch(`http://127.0.0.1:${port}/instances`, { headers: { 'x-dsh-agent-token': TOKEN } })
|
||||
const body = await res.text()
|
||||
console.error(` ${label} /instances = ${body.slice(0, 220)}`)
|
||||
const st = await fetch(`http://127.0.0.1:${port}/status/${target.id}`, { headers: { 'x-dsh-agent-token': TOKEN } })
|
||||
console.error(` ${label} /status = ${(await st.text()).slice(0, 220)}`)
|
||||
}
|
||||
tailLog('agent-w-2', 16)
|
||||
tailLog('manager', 8)
|
||||
}
|
||||
assert(newUrl !== '' && newUrl !== launchUrl, `迁移后 enter 应给**新** URL(旧 ${launchUrl} / 新 ${newUrl})`)
|
||||
assert(pageStatus === 200, `迁移后经新 URL 的实例页面应 200(实际 ${pageStatus})`)
|
||||
console.log('⑩ 迁移后 -> enter 返回新 token URL,页面 200(经 w-2)')
|
||||
|
||||
// ── ⑪ 观测面 ──────────────────────────────────────────────────────────
|
||||
const doctor = spawnSync(process.execPath, [CLI, 'doctor'], { cwd: repoRoot, env: { ...process.env, ...managerEnv }, encoding: 'utf8' })
|
||||
const status = spawnSync(process.execPath, [CLI, 'cluster', 'status'], { cwd: repoRoot, env: { ...process.env, ...managerEnv }, encoding: 'utf8' })
|
||||
console.log('⑪ dshs doctor -> rc=%d(0 = 无硬失败)', doctor.status ?? -1)
|
||||
console.log(String(status.stdout).split('\n').slice(0, 8).map((l) => ' ' + l).join('\n'))
|
||||
|
||||
// 收尾:停实例(避免留下 dsh 子进程)
|
||||
await json('/api/dsh/stop', { method: 'POST', cookie })
|
||||
|
||||
console.log('\nOK: 真实部署(2 个 worker agent 进程 + 1 个 Manager 进程 + 真 PG)端到端功能确认通过')
|
||||
console.log(' ✓ 用户流程 ✓ 文件面跨机 ✓ 真 dsh 拉起并可从公网侧取页面 ✓ 停止 ✓ 注册+迁移+迁移后复验 ✓ 观测面')
|
||||
console.log(' 页面片段:%s', pageSnippet.replace(/\s+/g, ' ').slice(0, 80))
|
||||
} finally {
|
||||
shutdownAll()
|
||||
await sleep(1500)
|
||||
}
|
||||
@@ -0,0 +1,246 @@
|
||||
/**
|
||||
* T08 S6 · 多 worker + 容量准入 + **计划内迁移**验证。
|
||||
*
|
||||
* 这一条是整套设计的落点:**实例可迁移**。它同时验证:
|
||||
* ① **容量准入**:`selectHost` 把"已用 + 预留 > 容量"的机排除掉 ⇒ 实例落到还有余量的那台;
|
||||
* ② **归属与实例一致**:`dsh_instances.host_id` 指向实例真正所在的那台;
|
||||
* ③ **迁移三步**(drain → 目标机拉起 → 归属原子更新):`host_id` 换台、`epoch` 单调 +1;
|
||||
* ④ **迁移后代理照常**:`endpointFor` 按新归属路由,页面仍 200;
|
||||
* ⑤ **数据不搬家也能用**:两台 worker **共享同一 dataRoot**(模拟共享存储 / 同路径基线)。
|
||||
*
|
||||
* 需要 PG(归属在 DB 里,两个 Manager/worker 共享):
|
||||
* CLUSTER_TEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node scripts/verify-cluster-migrate.mjs
|
||||
*/
|
||||
import { existsSync, mkdirSync, mkdtempSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { dirname, join } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
import pg from 'pg'
|
||||
import { buildServer } from '../lib/web/server.js'
|
||||
import { buildWorkerAgent, AGENT_TOKEN_HEADER } from '../lib/worker/agent.js'
|
||||
import { resolveConfig } from '../lib/config.js'
|
||||
import { hashPassword } from '../lib/web/auth.js'
|
||||
|
||||
function assert(condition, message) {
|
||||
if (!condition) throw new Error('ASSERT: ' + message)
|
||||
}
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
||||
|
||||
const PG_URL = process.env.CLUSTER_TEST_PG_URL
|
||||
if (PG_URL === undefined || PG_URL === '') {
|
||||
console.error('需要 CLUSTER_TEST_PG_URL')
|
||||
process.exit(2)
|
||||
}
|
||||
const TOKEN = 'verify-cluster-migrate-token'
|
||||
const here = dirname(fileURLToPath(import.meta.url))
|
||||
const fakeDsh = join(here, 'fake-dsh.mjs')
|
||||
/** 两台 worker **共享同一 dataRoot** = 模拟共享存储 / "所有 worker 同路径"的基线约定。 */
|
||||
const sharedRoot = mkdtempSync(join(tmpdir(), 'dsh-migrate-'))
|
||||
|
||||
// Manager 自己不承载实例(capacity=-1);两台 worker 声明 4096MB
|
||||
process.env.DSHS_CLUSTER_CAPACITY_MB = '-1'
|
||||
|
||||
let app
|
||||
const agents = []
|
||||
|
||||
async function resetPg() {
|
||||
const client = new pg.Client({ connectionString: PG_URL })
|
||||
await client.connect()
|
||||
await client.query('DELETE FROM dsh_instances')
|
||||
await client.query('DELETE FROM dsh_hosts')
|
||||
await client.query('DELETE FROM users')
|
||||
await client.end()
|
||||
}
|
||||
|
||||
async function startAgent(hostId) {
|
||||
const config = resolveConfig({
|
||||
port: 0,
|
||||
dbPath: ':memory:',
|
||||
dataRoot: sharedRoot,
|
||||
dshCommand: [process.execPath, fakeDsh],
|
||||
clusterHostId: hostId,
|
||||
})
|
||||
const agent = buildWorkerAgent(config, {
|
||||
hostId,
|
||||
token: TOKEN,
|
||||
port: 0,
|
||||
host: '127.0.0.1',
|
||||
instanceHost: '127.0.0.1',
|
||||
logLevel: 'warn',
|
||||
})
|
||||
await agent.app.listen({ host: '127.0.0.1', port: 0 })
|
||||
agents.push(agent) // 整个 handle:收尾要用 stop() 收实例
|
||||
const url = `http://127.0.0.1:${agent.app.server.address().port}`
|
||||
const call = async (path, { method = 'GET', body } = {}) => {
|
||||
const res = await fetch(url + path, {
|
||||
method,
|
||||
headers: { [AGENT_TOKEN_HEADER]: TOKEN, ...(body ? { 'content-type': 'application/json' } : {}) },
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
})
|
||||
const text = await res.text()
|
||||
return { status: res.status, body: text ? JSON.parse(text) : null }
|
||||
}
|
||||
/** 该机上的实例数(对账口径)。 */
|
||||
const instanceCount = async () => (await call('/instances')).body.instances.length
|
||||
return { hostId, url, call, instanceCount }
|
||||
}
|
||||
|
||||
try {
|
||||
await resetPg()
|
||||
|
||||
// ── 0) 两台 worker ────────────────────────────────────────────────────
|
||||
const a = await startAgent('w-a')
|
||||
const b = await startAgent('w-b')
|
||||
console.log('worker -> w-a %s / w-b %s(共享 dataRoot)', a.url, b.url)
|
||||
|
||||
// ── 1) Manager(默认 agent 指 w-a;自己 capacity=-1 不承载)─────────────
|
||||
app = await buildServer(
|
||||
resolveConfig({
|
||||
port: 0,
|
||||
dbUrl: PG_URL,
|
||||
dataRoot: sharedRoot,
|
||||
deployMode: 'cluster',
|
||||
clusterAgentUrl: a.url,
|
||||
clusterAgentToken: TOKEN,
|
||||
clusterInstanceHost: '127.0.0.1',
|
||||
clusterHostId: 'm-1',
|
||||
clusterWorkerDataRoot: sharedRoot,
|
||||
}),
|
||||
)
|
||||
await app.listen({ port: 0 })
|
||||
const base = `http://127.0.0.1:${app.server.address().port}`
|
||||
|
||||
// 注册两台 worker(join 脚本走的就是这个 API)
|
||||
await app.db.upsertDshHost({ id: 'w-a', endpoint: a.url, agentToken: TOKEN, capacityMb: 4096 })
|
||||
await app.db.upsertDshHost({ id: 'w-b', endpoint: b.url, agentToken: TOKEN, capacityMb: 4096 })
|
||||
// 把 w-b 的已用水位抬高到"再来一个实例就超" ⇒ 用来验证**准入拒绝**
|
||||
await app.db.setDshHostStatus('w-b', 'up', 3800, Date.now())
|
||||
|
||||
// admin 账号(迁移 API 需要)
|
||||
await app.db.createUser({
|
||||
id: 'admin-1',
|
||||
username: 'root',
|
||||
passHash: await hashPassword('rootpass123'),
|
||||
role: 'admin',
|
||||
homeDir: '/tmp/admin-home',
|
||||
})
|
||||
await app.db.createUser({
|
||||
id: 'u1',
|
||||
username: 'carol',
|
||||
passHash: await hashPassword('carolpass123'),
|
||||
role: 'active',
|
||||
homeDir: '/tmp/u1-home',
|
||||
})
|
||||
await app.userFs.initUserRoot('u1')
|
||||
// 门户流程里 folder 是用户从「我的文件」里挑的**已存在**目录 ⇒ 这里先建出来
|
||||
mkdirSync(join(sharedRoot, 'users', 'u1', 'ws', 'proj'), { recursive: true })
|
||||
|
||||
const json = async (path, { method = 'GET', body, cookie } = {}) => {
|
||||
const res = await fetch(base + path, {
|
||||
method,
|
||||
headers: { ...(body ? { 'content-type': 'application/json' } : {}), ...(cookie ? { cookie } : {}) },
|
||||
body: body ? JSON.stringify(body) : undefined,
|
||||
})
|
||||
const text = await res.text()
|
||||
return { status: res.status, body: text ? JSON.parse(text) : null, setCookie: res.headers.get('set-cookie') }
|
||||
}
|
||||
|
||||
// ── 2) 容量准入:w-b 水位高 ⇒ 必须落到 w-a ────────────────────────────
|
||||
const c = await json('/api/auth/login', { method: 'POST', body: { username: 'carol', password: 'carolpass123' } })
|
||||
assert(c.status === 200, 'user login')
|
||||
const userCookie = c.setCookie.split(';')[0]
|
||||
|
||||
let r = await json('/api/dsh/launch', { method: 'POST', cookie: userCookie, body: { folder: 'proj' } })
|
||||
assert(r.status === 200, `launch 经远端成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
|
||||
let row = await app.db.findUserInstance('u1', 'main')
|
||||
assert(row.hostId === 'w-a', `容量准入应选 w-a(w-b 已 3800+512>4096);实际 ${row.hostId}`)
|
||||
assert(row.epoch === 1, `首次抢占 epoch=1(实际 ${row.epoch})`)
|
||||
assert((await a.instanceCount()) === 1, 'w-a 上有 1 个实例')
|
||||
assert((await b.instanceCount()) === 0, 'w-b 上 0 个实例')
|
||||
console.log('① 容量准入 -> 落到 w-a(w-b 因水位被排除),host_id=w-a epoch=1')
|
||||
|
||||
// 代理照常
|
||||
let proxyText
|
||||
for (let i = 0; i < 20; i += 1) {
|
||||
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie: userCookie } })
|
||||
if (res.status === 200) {
|
||||
proxyText = await res.text()
|
||||
break
|
||||
}
|
||||
await sleep(100)
|
||||
}
|
||||
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), '迁移前代理 200(经 w-a)')
|
||||
console.log(' 迁移前代理 -> 200(经 w-a)')
|
||||
|
||||
// 顺便在用户工作区放个文件(迁移后要还在 —— 共享存储场景)
|
||||
await json('/api/fs/upload', {
|
||||
method: 'POST',
|
||||
cookie: userCookie,
|
||||
body: { path: 'proj', name: 'keep.txt', data: Buffer.from('survives migration').toString('base64') },
|
||||
})
|
||||
|
||||
// ── 3) 迁移到 w-b ─────────────────────────────────────────────────────
|
||||
const adm = await json('/api/auth/login', { method: 'POST', body: { username: 'root', password: 'rootpass123' } })
|
||||
assert(adm.status === 200, 'admin login')
|
||||
const adminCookie = adm.setCookie.split(';')[0]
|
||||
|
||||
r = await json('/api/admin/users/u1/dsh/migrate', {
|
||||
method: 'POST',
|
||||
cookie: adminCookie,
|
||||
body: { targetHost: 'w-b' },
|
||||
})
|
||||
assert(r.status === 200, `迁移成功(实际 ${r.status} ${JSON.stringify(r.body)})`)
|
||||
assert(r.body.from === 'w-a' && r.body.to === 'w-b', `迁移方向 w-a→w-b(实际 ${JSON.stringify(r.body)})`)
|
||||
assert(r.body.epoch === 2, `epoch 应 +1 到 2(实际 ${r.body.epoch})`)
|
||||
row = await app.db.findUserInstance('u1', 'main')
|
||||
assert(row.hostId === 'w-b' && row.epoch === 2, '归属已原子更新到 w-b / epoch=2')
|
||||
assert((await a.instanceCount()) === 0, 'w-a 上实例已停(drain 生效)')
|
||||
assert((await b.instanceCount()) === 1, 'w-b 上有 1 个实例')
|
||||
console.log('② 迁移 -> w-a → w-b,host_id=w-b epoch=%d,源机实例已停', r.body.epoch)
|
||||
|
||||
// 迁移后代理照常(按新归属路由到 w-b)
|
||||
proxyText = undefined
|
||||
for (let i = 0; i < 20; i += 1) {
|
||||
const res = await fetch(`${base}/u/u1/dsh/hello`, { headers: { cookie: userCookie } })
|
||||
if (res.status === 200) {
|
||||
proxyText = await res.text()
|
||||
break
|
||||
}
|
||||
await sleep(100)
|
||||
}
|
||||
assert(proxyText !== undefined && proxyText.includes('fake-dsh'), '迁移后代理 200(经 w-b)')
|
||||
|
||||
// ── 4) 数据还在(共享存储)────────────────────────────────────────────
|
||||
const keptPath = join(sharedRoot, 'users', 'u1', 'ws', 'proj', 'keep.txt')
|
||||
assert(existsSync(keptPath), `迁移后文件仍在:${keptPath}`)
|
||||
const dl = await fetch(`${base}/api/fs/download?path=proj/keep.txt`, { headers: { cookie: userCookie } })
|
||||
assert(dl.status === 200 && (await dl.text()) === 'survives migration', '迁移后仍能下载到原文')
|
||||
console.log('③ 迁移后 -> 代理 200(经 w-b)、文件可读(数据不搬家)')
|
||||
|
||||
// ── 5) 已在该机 + 目标机不存在 ⇒ 明确报错(不静默)────────────────────
|
||||
r = await json('/api/admin/users/u1/dsh/migrate', { method: 'POST', cookie: adminCookie, body: { targetHost: 'w-b' } })
|
||||
assert(r.status === 409 && r.body.error === 'already_there', `重复迁移应 409 already_there(实际 ${r.status})`)
|
||||
r = await json('/api/admin/users/u1/dsh/migrate', { method: 'POST', cookie: adminCookie, body: { targetHost: 'nope' } })
|
||||
assert(r.status === 404 && r.body.error === 'unknown_host', `未知目标机应 404(实际 ${r.status})`)
|
||||
console.log('④ 边界 -> already_there / unknown_host 都明确报错')
|
||||
|
||||
console.log('\nOK: 多 worker + 容量准入 + 迁移通过')
|
||||
console.log(' ✓ 容量准入 ✓ 归属与实例一致 ✓ 迁移(drain→拉起→epoch+1) ✓ 迁移后代理/文件正常')
|
||||
} finally {
|
||||
// ⚠️ 收尾必须**停掉还活着的实例**:否则 fake-dsh 子进程会继承 stdout,
|
||||
// 管道永不关闭 ⇒ ssh / CI 会一直挂在这里(2026-09-15 实测踩到)。
|
||||
try {
|
||||
await app?.supervisor?.stop('u1')
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
await app?.close()
|
||||
for (const h of agents) await h?.stop()
|
||||
await sleep(500)
|
||||
try {
|
||||
rmSync(sharedRoot, { recursive: true, force: true })
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
}
|
||||
+214
@@ -23,6 +23,9 @@ const HELP = `dshs — DSH server login orchestrator
|
||||
Usage:
|
||||
dshs [options] start the server
|
||||
dshs bootstrap-admin [options] create the first admin
|
||||
dshs worker [options] worker agent(cluster 模式:承载本机实例)
|
||||
dshs doctor [--json] 单机自检(环境/隔离/存储/DB;非 0 退出 = 有硬失败)
|
||||
dshs cluster status [--json] 全集群一屏(worker 目录 + 归属 + 过期租约)
|
||||
|
||||
Server options:
|
||||
--port <n> Bind port (0 = ephemeral). Default 3080.
|
||||
@@ -65,6 +68,138 @@ function toOverrides(values: ParsedValues): ConfigOverrides {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* `dshs doctor`:**单机自检**(T08 S7;设计 §15.5)。
|
||||
*
|
||||
* 把"装机/排障要逐条手查"的东西固化成一条命令:环境 → 隔离能力 → 存储 → DB。
|
||||
* **退出码非 0 = 有硬失败**(可直接用于 join 脚本的门禁);warn 不影响退出码。
|
||||
*/
|
||||
async function doctorCmd(args: string[]): Promise<void> {
|
||||
const { values } = parseArgs({ args, options: { json: { type: 'boolean' } } })
|
||||
const { execFileSync } = await import('node:child_process')
|
||||
const { statfsSync, accessSync, constants } = await import('node:fs')
|
||||
const config = resolveConfig({})
|
||||
const lines: Array<{ level: 'ok' | 'warn' | 'fail'; item: string; detail: string }> = []
|
||||
const add = (level: 'ok' | 'warn' | 'fail', item: string, detail: string): void => {
|
||||
lines.push({ level, item, detail })
|
||||
}
|
||||
|
||||
// ── 环境 ───────────────────────────────────────────────────────────────
|
||||
add('ok', 'node', process.version)
|
||||
let cgroup = 'unknown'
|
||||
try {
|
||||
cgroup = statfsSync('/sys/fs/cgroup').type === 0x63677270 ? 'v2' : 'v1'
|
||||
} catch {
|
||||
cgroup = 'unknown'
|
||||
}
|
||||
add(cgroup === 'unknown' ? 'warn' : 'ok', 'cgroup', cgroup)
|
||||
let swap = ''
|
||||
try {
|
||||
// ⚠️ /proc/swaps 第一行是表头 ⇒ 要 NR>1,否则会把 "Filename" 当成设备名打出来
|
||||
swap = execFileSync('/usr/bin/awk', ['NR>1 && NF>0 {print $1}', '/proc/swaps'], { encoding: 'utf8' })
|
||||
.trim()
|
||||
.replace(/\n+/g, ' ')
|
||||
} catch {
|
||||
swap = ''
|
||||
}
|
||||
add(swap === '' ? 'ok' : 'warn', 'swap', swap === '' ? '未启用' : `已启用(${swap})—— 实例超限会先换出而非被 OOM kill`)
|
||||
for (const bin of ['bwrap', 'setpriv', 'systemd-run', 'nft', 'dsh']) {
|
||||
const found = execFileSync('/usr/bin/which', [bin], { encoding: 'utf8' }).trim()
|
||||
add(found === '' ? 'fail' : 'ok', bin, found === '' ? '缺失' : found)
|
||||
}
|
||||
let bwrapVersion = ''
|
||||
try {
|
||||
bwrapVersion = execFileSync('bwrap', ['--version'], { encoding: 'utf8' }).trim()
|
||||
} catch {
|
||||
bwrapVersion = ''
|
||||
}
|
||||
const minor = /bubblewrap (\d+)\.(\d+)/.exec(bwrapVersion)
|
||||
if (minor !== null && Number(minor[2]) < 5) {
|
||||
add('warn', 'bwrap 版本', `${bwrapVersion} —— **低于 0.5:不支持 --perms**(要改挂载点权限只能用 --tmpfs)`)
|
||||
} else if (bwrapVersion !== '') {
|
||||
add('ok', 'bwrap 版本', bwrapVersion)
|
||||
}
|
||||
|
||||
// ── 隔离前提 ───────────────────────────────────────────────────────────
|
||||
try {
|
||||
const uid = execFileSync('setpriv', ['--reuid', '100001', '--regid', '100001', '--clear-groups', '--', 'id', '-u'], {
|
||||
encoding: 'utf8',
|
||||
}).trim()
|
||||
add('ok', 'setpriv 降权', `可用(uid=${uid})`)
|
||||
} catch {
|
||||
add('fail', 'setpriv 降权', '失败(需要 root 或 CAP_SETUID)')
|
||||
}
|
||||
|
||||
// ── 存储 ───────────────────────────────────────────────────────────────
|
||||
try {
|
||||
accessSync(config.dataRoot, constants.W_OK)
|
||||
add('ok', 'dataRoot 可写', config.dataRoot)
|
||||
} catch {
|
||||
add('fail', 'dataRoot 可写', `${config.dataRoot} 不可写`)
|
||||
}
|
||||
|
||||
// ── DB ─────────────────────────────────────────────────────────────────
|
||||
try {
|
||||
const { createDbAdapter } = await import('./db/index.js')
|
||||
const db = await createDbAdapter(config)
|
||||
const hosts = await db.listDshHosts()
|
||||
await db.close()
|
||||
add('ok', 'DB', `${config.dbUrl === undefined ? `sqlite ${config.dbPath}` : 'postgres'}(dsh_hosts ${hosts.length} 条)`)
|
||||
} catch (err) {
|
||||
add('fail', 'DB', err instanceof Error ? err.message : String(err))
|
||||
}
|
||||
|
||||
const failures = lines.filter((l) => l.level === 'fail').length
|
||||
if (values.json === true) {
|
||||
process.stdout.write(JSON.stringify({ ok: failures === 0, checks: lines }, null, 2) + '\n')
|
||||
} else {
|
||||
for (const l of lines) {
|
||||
const mark = l.level === 'ok' ? '✓' : l.level === 'warn' ? '!' : '✗'
|
||||
process.stdout.write(`${mark} ${l.item.padEnd(16)} ${l.detail}\n`)
|
||||
}
|
||||
process.stdout.write(`\n${failures === 0 ? 'OK:无硬失败' : `${failures} 项硬失败`}\n`)
|
||||
}
|
||||
if (failures > 0) process.exit(2)
|
||||
}
|
||||
|
||||
/**
|
||||
* `dshs cluster status`:**全集群一屏**(T08 S7;设计 §15.5)。
|
||||
* 读的是**控制面 DB**(Manager 侧运行),输出 worker 目录 + 实例归属 + 过期租约。
|
||||
*/
|
||||
async function clusterStatusCmd(args: string[]): Promise<void> {
|
||||
const { values } = parseArgs({ args, options: { json: { type: 'boolean' } } })
|
||||
const config = resolveConfig({})
|
||||
const { createDbAdapter } = await import('./db/index.js')
|
||||
const db = await createDbAdapter(config)
|
||||
try {
|
||||
const [hosts, expired] = await Promise.all([
|
||||
db.listDshHosts(),
|
||||
db.listExpiredInstanceLeases(Date.now()),
|
||||
])
|
||||
const byHost = new Map<string, number>()
|
||||
for (const h of hosts) byHost.set(h.id, (await db.listInstancesByHost(h.id)).length)
|
||||
if (values.json === true) {
|
||||
process.stdout.write(JSON.stringify({ deployMode: config.deployMode, hosts, expired }, null, 2) + '\n')
|
||||
return
|
||||
}
|
||||
process.stdout.write(`deployMode=${config.deployMode} db=${config.dbUrl === undefined ? config.dbPath : 'postgres'}\n\n`)
|
||||
process.stdout.write('WORKER 状态 容量(MB) 已用 实例 最后心跳\n')
|
||||
for (const h of hosts) {
|
||||
const hb = h.lastHeartbeat === null ? '从未' : new Date(h.lastHeartbeat).toISOString().replace('T', ' ').slice(0, 19)
|
||||
process.stdout.write(
|
||||
`${h.id.padEnd(26)} ${h.status.padEnd(8)} ${String(h.capacityMb).padStart(8)} ${String(h.usedMb).padStart(6)} ` +
|
||||
`${String(byHost.get(h.id) ?? 0).padStart(6)} ${hb}\n`,
|
||||
)
|
||||
}
|
||||
process.stdout.write(`\n租约已过期(需人工确认后才可接管,见 R9):${expired.length} 个\n`)
|
||||
for (const inst of expired) {
|
||||
process.stdout.write(` ${inst.userId} host=${inst.hostId ?? '-'} epoch=${inst.epoch} 过期于 ${new Date(inst.leaseUntil).toISOString()}\n`)
|
||||
}
|
||||
} finally {
|
||||
await db.close()
|
||||
}
|
||||
}
|
||||
|
||||
async function bootstrapAdmin(args: string[]): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
args,
|
||||
@@ -140,6 +275,69 @@ async function runServer(args: string[]): Promise<void> {
|
||||
process.on('SIGTERM', () => void shutdown('SIGTERM'))
|
||||
}
|
||||
|
||||
/**
|
||||
* 运行 **worker agent**(T08 S3;设计 §11.2)。
|
||||
*
|
||||
* 它是 Worker 上唯一的"被拨入口":把本机的实例生命周期(launch/stop/status/endpoint/fence)
|
||||
* 暴露给 Manager。**不连控制面 DB** —— 凭据(apiKey)与 uid 由 Manager 在 launch 时投递,
|
||||
* 只存内存(与 k8s 用 per-user Secret 同一思路)。**不是"不许有数据库"**:插件业务数据在
|
||||
* 实例 home 里、由实例自己读写(设计 §1.3 数据分层)。
|
||||
*/
|
||||
async function runWorker(args: string[]): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
args,
|
||||
options: {
|
||||
port: { type: 'string' },
|
||||
host: { type: 'string' },
|
||||
'host-id': { type: 'string' },
|
||||
token: { type: 'string' },
|
||||
'instance-host': { type: 'string' },
|
||||
'log-level': { type: 'string' },
|
||||
help: { type: 'boolean', short: 'h' },
|
||||
},
|
||||
})
|
||||
if (values.help === true) {
|
||||
process.stdout.write(
|
||||
'usage: dshs worker --token <secret> [--port 9000] [--host 0.0.0.0]\n' +
|
||||
' [--host-id <id>] [--instance-host <addr>]\n' +
|
||||
' 密钥也可用 DSHS_CLUSTER_AGENT_TOKEN;host-id 默认取主机名。\n',
|
||||
)
|
||||
return
|
||||
}
|
||||
const token =
|
||||
(typeof values.token === 'string' ? values.token : undefined) ?? process.env.DSHS_CLUSTER_AGENT_TOKEN
|
||||
if (token === undefined || token === '') {
|
||||
console.error('worker requires --token or DSHS_CLUSTER_AGENT_TOKEN')
|
||||
process.exit(2)
|
||||
}
|
||||
const config = resolveConfig({
|
||||
logLevel: typeof values['log-level'] === 'string' ? values['log-level'] : undefined,
|
||||
clusterHostId: typeof values['host-id'] === 'string' ? values['host-id'] : undefined,
|
||||
clusterInstanceHost: typeof values['instance-host'] === 'string' ? values['instance-host'] : undefined,
|
||||
})
|
||||
const { buildWorkerAgent } = await import('./worker/agent.js')
|
||||
const host = typeof values.host === 'string' ? values.host : '0.0.0.0'
|
||||
const port = Number(typeof values.port === 'string' ? values.port : 9000)
|
||||
const agent = buildWorkerAgent(config, {
|
||||
hostId: config.clusterHostId,
|
||||
token,
|
||||
port,
|
||||
host,
|
||||
instanceHost: config.clusterInstanceHost,
|
||||
logLevel: config.logLevel,
|
||||
})
|
||||
await agent.app.listen({ host, port })
|
||||
agent.app.log.info(`worker agent listening on http://${host}:${port} (hostId ${config.clusterHostId})`)
|
||||
|
||||
const shutdown = async (signal: string): Promise<void> => {
|
||||
agent.app.log.info(`received ${signal}, shutting down`)
|
||||
await agent.stop()
|
||||
process.exit(0)
|
||||
}
|
||||
process.on('SIGINT', () => void shutdown('SIGINT'))
|
||||
process.on('SIGTERM', () => void shutdown('SIGTERM'))
|
||||
}
|
||||
|
||||
async function uidForUserCmd(args: string[]): Promise<void> {
|
||||
const { values, positionals } = parseArgs({
|
||||
args,
|
||||
@@ -166,6 +364,22 @@ async function main(): Promise<void> {
|
||||
await bootstrapAdmin(rest)
|
||||
return
|
||||
}
|
||||
if (first === 'worker') {
|
||||
await runWorker(rest)
|
||||
return
|
||||
}
|
||||
if (first === 'doctor') {
|
||||
await doctorCmd(rest)
|
||||
return
|
||||
}
|
||||
if (first === 'cluster') {
|
||||
if (rest[0] === 'status') {
|
||||
await clusterStatusCmd(rest.slice(1))
|
||||
return
|
||||
}
|
||||
process.stderr.write('usage: dshs cluster status [--json]\n')
|
||||
process.exit(2)
|
||||
}
|
||||
if (first === 'uid-for-user') {
|
||||
await uidForUserCmd(rest)
|
||||
return
|
||||
|
||||
+28
-3
@@ -15,7 +15,7 @@ export type IsolationMode = 'soft' | 'account'
|
||||
|
||||
/** Deployment mode. `local` = single-host child_process (setuid/iptables);
|
||||
* `k8s` = multi-replica control plane spawning per-user DSH Pods via the K8s API. */
|
||||
export type DeployMode = 'local' | 'k8s'
|
||||
export type DeployMode = 'local' | 'k8s' | 'cluster'
|
||||
|
||||
/** Resolved, immutable runtime configuration. */
|
||||
export interface ServerConfig {
|
||||
@@ -96,6 +96,21 @@ export interface ServerConfig {
|
||||
k8sServiceAccount: string
|
||||
/** This replica's identity for leader election (POD_NAME, else hostname). */
|
||||
podName: string
|
||||
// ── cluster 模式(T08 S3/S4;设计 §1.1)────────────────────────────────
|
||||
/** 本机在 `dsh_hosts.id` 里的标识(`deployMode=cluster` 时必填语义)。 */
|
||||
clusterHostId: string
|
||||
/** 本机 worker agent 的**基址**(Manager 侧用它投递实例操作),如 `http://127.0.0.1:9000`。 */
|
||||
clusterAgentUrl: string
|
||||
/** 与 agent 约定的共享密钥(仅内网 + nft 白名单)。 */
|
||||
clusterAgentToken: string
|
||||
/** agent 返回给 Manager 做代理的实例地址(同机 1a = `127.0.0.1`)。 */
|
||||
clusterInstanceHost: string
|
||||
/**
|
||||
* **worker 上**的 dataRoot(T08 S5)。
|
||||
* 空 = 与本地 `dataRoot` 相同(1a 形态)。多机部署必须显式配置 —— 而且是**基线约定**:
|
||||
* 所有 worker 的 dataRoot 必须是同一个绝对路径(同镜像即可满足,设计 §14.3)。
|
||||
*/
|
||||
clusterWorkerDataRoot: string
|
||||
}
|
||||
|
||||
/** Untyped overrides collected from argv / env. */
|
||||
@@ -137,6 +152,11 @@ export interface ConfigOverrides {
|
||||
egressCidrs?: string[]
|
||||
k8sServiceAccount?: string
|
||||
podName?: string
|
||||
clusterHostId?: string
|
||||
clusterAgentUrl?: string
|
||||
clusterAgentToken?: string
|
||||
clusterInstanceHost?: string
|
||||
clusterWorkerDataRoot?: string
|
||||
}
|
||||
|
||||
const DEFAULT_HOST = '127.0.0.1'
|
||||
@@ -223,8 +243,8 @@ function toIsolationMode(value: string | undefined): IsolationMode | undefined {
|
||||
function toDeployMode(value: string | undefined): DeployMode | undefined {
|
||||
if (value === undefined) return undefined
|
||||
const normalized = value.trim().toLowerCase()
|
||||
if (normalized === 'local' || normalized === 'k8s') return normalized
|
||||
throw new Error(`invalid deploy mode "${value}" (expected "local" or "k8s")`)
|
||||
if (normalized === 'local' || normalized === 'k8s' || normalized === 'cluster') return normalized
|
||||
throw new Error(`invalid deploy mode "${value}" (expected "local", "k8s" or "cluster")`)
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -327,5 +347,10 @@ export function resolveConfig(overrides: ConfigOverrides = {}): ServerConfig {
|
||||
k8sServiceAccount:
|
||||
overrides.k8sServiceAccount ?? process.env.DSHS_K8S_SERVICE_ACCOUNT ?? DEFAULT_K8S_SERVICE_ACCOUNT,
|
||||
podName: overrides.podName ?? process.env.POD_NAME ?? hostname(),
|
||||
clusterHostId: overrides.clusterHostId ?? process.env.DSHS_CLUSTER_HOST_ID ?? hostname(),
|
||||
clusterAgentUrl: overrides.clusterAgentUrl ?? process.env.DSHS_CLUSTER_AGENT_URL ?? '',
|
||||
clusterAgentToken: overrides.clusterAgentToken ?? process.env.DSHS_CLUSTER_AGENT_TOKEN ?? '',
|
||||
clusterInstanceHost: overrides.clusterInstanceHost ?? process.env.DSHS_CLUSTER_INSTANCE_HOST ?? '127.0.0.1',
|
||||
clusterWorkerDataRoot: overrides.clusterWorkerDataRoot ?? process.env.DSHS_CLUSTER_WORKER_DATA_ROOT ?? '',
|
||||
}
|
||||
}
|
||||
@@ -7,12 +7,15 @@
|
||||
|
||||
import type {
|
||||
BusinessPlugin,
|
||||
ClaimResult,
|
||||
CredentialKey,
|
||||
CredentialKeyMeta,
|
||||
CredentialLandingRow,
|
||||
CreateSessionInput,
|
||||
CreateUserInput,
|
||||
Domain,
|
||||
DshHost,
|
||||
DshHostStatus,
|
||||
DshInstance,
|
||||
DshInstanceRole,
|
||||
DshInstanceStatus,
|
||||
@@ -20,6 +23,7 @@ import type {
|
||||
SessionRow,
|
||||
SessionUser,
|
||||
UpsertBusinessPluginInput,
|
||||
UpsertDshHostInput,
|
||||
UpsertDshInstanceInput,
|
||||
User,
|
||||
UserRole,
|
||||
@@ -104,6 +108,41 @@ export interface DbAdapter {
|
||||
): Promise<boolean>
|
||||
deleteInstance(id: string): Promise<boolean>
|
||||
deleteUserInstances(userId: string): Promise<void>
|
||||
// ── 集群化:worker 注册表 + 实例归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)──
|
||||
// ⚠️ local 模式**不写**这些表(`LocalSpawner` 靠进程内 Map + 单机互斥),
|
||||
// 所以这些方法在单机路径上恒为"空/未认领",不影响现有行为。
|
||||
/** 注册/更新一台 worker(join 幂等:同 id 重复执行 = 更新)。 */
|
||||
upsertDshHost(input: UpsertDshHostInput): Promise<DshHost>
|
||||
findDshHost(id: string): Promise<DshHost | undefined>
|
||||
listDshHosts(): Promise<DshHost[]>
|
||||
/** 心跳/状态上报:可只改状态,或同时带上容量水位与心跳时间。 */
|
||||
setDshHostStatus(
|
||||
id: string,
|
||||
status: DshHostStatus,
|
||||
usedMb?: number,
|
||||
heartbeatAt?: number,
|
||||
): Promise<boolean>
|
||||
/**
|
||||
* **原子抢占**某用户的 main 实例归属(承重墙,见设计 §3.2)。
|
||||
* 仅当"无人持有 **或** 租约已过期"才成功;成功时 `epoch` +1(fencing)。
|
||||
* 返回 `ok:false` = 有人在管 ⇒ 调用方**退让**(不是接管)。
|
||||
*/
|
||||
claimInstance(
|
||||
userId: string,
|
||||
hostId: string,
|
||||
ttlMs: number,
|
||||
meta?: { folder?: string; patch?: string },
|
||||
): Promise<ClaimResult>
|
||||
/** 续租。**必须带 epoch**:不匹配说明已被他人抢占 ⇒ 本次续租失败(fencing)。 */
|
||||
renewInstanceLease(userId: string, hostId: string, epoch: number, ttlMs: number): Promise<boolean>
|
||||
/** 主动释放(停实例时)。同样带 epoch 校验,避免误清他人的归属。 */
|
||||
releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean>
|
||||
/** 钉住归属(首次触达工作区时用):只写 `host_id`,不动 epoch/租约。 */
|
||||
pinInstanceHost(userId: string, hostId: string): Promise<void>
|
||||
/** 租约已过期、但仍标着归属的实例 —— 供巡检/自愈(**不代表可以立即接管**,见 R9)。 */
|
||||
listExpiredInstanceLeases(now: number): Promise<DshInstance[]>
|
||||
/** 某 worker 上的全部实例 —— 对账用**一次拿回整机**(替代逐用户查询)。 */
|
||||
listInstancesByHost(hostId: string): Promise<DshInstance[]>
|
||||
// lifecycle
|
||||
close(): Promise<void>
|
||||
}
|
||||
+156
-1
@@ -11,20 +11,25 @@ import type { DbAdapter } from './adapter.js'
|
||||
import { mapPgError } from './errors.js'
|
||||
import { runPgMigrations } from './schema.js'
|
||||
import {
|
||||
clusterInstanceId,
|
||||
toBusinessPlugin,
|
||||
toDomain,
|
||||
toDshHost,
|
||||
toDshInstance,
|
||||
toPublicUser,
|
||||
toSession,
|
||||
toUser,
|
||||
toWorkspace,
|
||||
type BusinessPlugin,
|
||||
type ClaimResult,
|
||||
type CredentialKey,
|
||||
type CredentialKeyMeta,
|
||||
type CredentialLandingRow,
|
||||
type CreateSessionInput,
|
||||
type CreateUserInput,
|
||||
type Domain,
|
||||
type DshHost,
|
||||
type DshHostStatus,
|
||||
type DshInstance,
|
||||
type DshInstanceRole,
|
||||
type DshInstanceStatus,
|
||||
@@ -32,6 +37,7 @@ import {
|
||||
type SessionRow,
|
||||
type SessionUser,
|
||||
type UpsertBusinessPluginInput,
|
||||
type UpsertDshHostInput,
|
||||
type UpsertDshInstanceInput,
|
||||
type User,
|
||||
type UserRole,
|
||||
@@ -47,8 +53,10 @@ types.setTypeParser(20, (value: string) => Number(value))
|
||||
const USER_COLS = 'id, username, pass_hash, role, home_dir, api_key_ref, created_at, approved_by, uid'
|
||||
const DOMAIN_COLS = 'id, user_id, domain, verified, nginx_config, updated_at'
|
||||
const BUSINESS_PLUGIN_COLS = 'id, name, description, version, tgz_path, file_size, uploaded_by, created_at, updated_at'
|
||||
const HOST_COLS = 'id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat'
|
||||
const INSTANCE_COLS =
|
||||
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch'
|
||||
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch, '
|
||||
+ 'host_id, epoch, heartbeat_at, lease_until' // v7 集群化归属/租约(T08 S2)—— 漏了它们会让 hostId 恒为 null
|
||||
|
||||
/** Run `fn` on a dedicated client inside a BEGIN/COMMIT/ROLLBACK transaction. */
|
||||
export async function withTx<T>(pool: Pool, fn: (client: PoolClient) => Promise<T>): Promise<T> {
|
||||
@@ -587,6 +595,153 @@ export class PgAdapter implements DbAdapter {
|
||||
await this.pool.query('DELETE FROM dsh_instances WHERE user_id = $1', [userId])
|
||||
}
|
||||
|
||||
// ── 集群化:worker 注册表 + 归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)────────
|
||||
// 与 `repo.ts` 的同名 SQLite 实现**逐条对齐**(两套实现并存是本库既有事实,
|
||||
// 见档案 19 §C8):任何 schema/语义变更都要**两侧同改**,否则切库时才炸。
|
||||
|
||||
async upsertDshHost(input: UpsertDshHostInput): Promise<DshHost> {
|
||||
try {
|
||||
const { rows } = await this.pool.query(
|
||||
`INSERT INTO dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat)
|
||||
VALUES ($1, $2, $3, $4, 0, $5, NULL)
|
||||
ON CONFLICT(id) DO UPDATE SET
|
||||
endpoint = excluded.endpoint,
|
||||
agent_token = excluded.agent_token,
|
||||
capacity_mb = excluded.capacity_mb,
|
||||
status = excluded.status
|
||||
RETURNING id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat`,
|
||||
[input.id, input.endpoint, input.agentToken, input.capacityMb, input.status ?? 'up'],
|
||||
)
|
||||
return toDshHost(rows[0] as Record<string, unknown>)
|
||||
} catch (e) {
|
||||
mapPgError(e)
|
||||
}
|
||||
}
|
||||
|
||||
async findDshHost(id: string): Promise<DshHost | undefined> {
|
||||
const { rows } = await this.pool.query(
|
||||
`SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = $1`,
|
||||
[id],
|
||||
)
|
||||
return rows.length > 0 ? toDshHost(rows[0] as Record<string, unknown>) : undefined
|
||||
}
|
||||
|
||||
async listDshHosts(): Promise<DshHost[]> {
|
||||
const { rows } = await this.pool.query(`SELECT ${HOST_COLS} FROM dsh_hosts ORDER BY id ASC`)
|
||||
return rows.map((row) => toDshHost(row as Record<string, unknown>))
|
||||
}
|
||||
|
||||
async setDshHostStatus(
|
||||
id: string,
|
||||
status: DshHostStatus,
|
||||
usedMb?: number,
|
||||
heartbeatAt?: number,
|
||||
): Promise<boolean> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE dsh_hosts
|
||||
SET status = $1,
|
||||
used_mb = COALESCE($2, used_mb),
|
||||
last_heartbeat = COALESCE($3, last_heartbeat)
|
||||
WHERE id = $4`,
|
||||
[status, usedMb ?? null, heartbeatAt ?? null, id],
|
||||
)
|
||||
return (result.rowCount ?? 0) > 0
|
||||
}
|
||||
|
||||
/**
|
||||
* **原子抢占**(承重墙):PG 侧用 `UPDATE … RETURNING` —— 只有真正更新到行才返回行,
|
||||
* 比"先读后写"少一次竞态窗口(SQLite 侧用 `changes` 判定,语义等价)。
|
||||
*/
|
||||
async claimInstance(
|
||||
userId: string,
|
||||
hostId: string,
|
||||
ttlMs: number,
|
||||
meta?: { folder?: string; patch?: string },
|
||||
): Promise<ClaimResult> {
|
||||
const now = Date.now()
|
||||
const id = clusterInstanceId(userId)
|
||||
await this.pool.query(
|
||||
`INSERT INTO dsh_instances (id, user_id, role, status) VALUES ($1, $2, 'main', 'starting')
|
||||
ON CONFLICT(id) DO NOTHING`,
|
||||
[id, userId],
|
||||
)
|
||||
// folder/patch 一起落库:迁移要能复现启动参数(见 repo.ts 同名处注释)
|
||||
const res = await this.pool.query(
|
||||
`UPDATE dsh_instances
|
||||
SET host_id = $1, epoch = epoch + 1, heartbeat_at = $2, lease_until = $3,
|
||||
folder = COALESCE($4, folder), patch = COALESCE($5, patch)
|
||||
WHERE id = $6 AND (host_id IS NULL OR lease_until < $2)
|
||||
RETURNING epoch, lease_until`,
|
||||
[hostId, now, now + ttlMs, meta?.folder ?? null, meta?.patch ?? null, id],
|
||||
)
|
||||
if (res.rows.length > 0) {
|
||||
const row = res.rows[0] as { epoch: number; lease_until: number }
|
||||
return { ok: true, epoch: row.epoch, leaseUntil: row.lease_until }
|
||||
}
|
||||
const cur = await this.pool.query('SELECT host_id, lease_until FROM dsh_instances WHERE id = $1', [id])
|
||||
const row = cur.rows[0] as { host_id: string | null; lease_until: number } | undefined
|
||||
return { ok: false, holder: row?.host_id ?? null, leaseUntil: row?.lease_until ?? 0 }
|
||||
}
|
||||
|
||||
async renewInstanceLease(
|
||||
userId: string,
|
||||
hostId: string,
|
||||
epoch: number,
|
||||
ttlMs: number,
|
||||
): Promise<boolean> {
|
||||
const now = Date.now()
|
||||
const result = await this.pool.query(
|
||||
`UPDATE dsh_instances SET heartbeat_at = $1, lease_until = $2
|
||||
WHERE id = $3 AND host_id = $4 AND epoch = $5`,
|
||||
[now, now + ttlMs, clusterInstanceId(userId), hostId, epoch],
|
||||
)
|
||||
return (result.rowCount ?? 0) > 0
|
||||
}
|
||||
|
||||
/** 只清租约、**保留 host_id**(见 repo.ts 同名函数的长注释)。 */
|
||||
async releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE dsh_instances SET lease_until = 0
|
||||
WHERE id = $1 AND host_id = $2 AND epoch = $3`,
|
||||
[clusterInstanceId(userId), hostId, epoch],
|
||||
)
|
||||
return (result.rowCount ?? 0) > 0
|
||||
}
|
||||
|
||||
/** 钉住归属(首次触达工作区时用):只写 host_id。 */
|
||||
async pinInstanceHost(userId: string, hostId: string): Promise<void> {
|
||||
const now = Date.now()
|
||||
const id = clusterInstanceId(userId)
|
||||
await this.pool.query(
|
||||
`INSERT INTO dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
|
||||
VALUES ($1, $2, 'main', 'stopped', $3, 0, 0, 0)
|
||||
ON CONFLICT(id) DO NOTHING`,
|
||||
[id, userId, hostId],
|
||||
)
|
||||
await this.pool.query(
|
||||
`UPDATE dsh_instances SET host_id = $1 WHERE id = $2 AND (host_id IS NULL OR lease_until < $3)`,
|
||||
[hostId, id, now],
|
||||
)
|
||||
}
|
||||
|
||||
async listExpiredInstanceLeases(now: number): Promise<DshInstance[]> {
|
||||
const { rows } = await this.pool.query(
|
||||
`SELECT ${INSTANCE_COLS} FROM dsh_instances
|
||||
WHERE role = 'main' AND host_id IS NOT NULL AND lease_until < $1 AND status <> 'stopped'
|
||||
ORDER BY lease_until ASC`,
|
||||
[now],
|
||||
)
|
||||
return rows.map((row) => toDshInstance(row as Record<string, unknown>))
|
||||
}
|
||||
|
||||
async listInstancesByHost(hostId: string): Promise<DshInstance[]> {
|
||||
const { rows } = await this.pool.query(
|
||||
`SELECT ${INSTANCE_COLS} FROM dsh_instances WHERE host_id = $1 ORDER BY user_id ASC`,
|
||||
[hostId],
|
||||
)
|
||||
return rows.map((row) => toDshInstance(row as Record<string, unknown>))
|
||||
}
|
||||
|
||||
async close(): Promise<void> {
|
||||
await this.pool.end()
|
||||
}
|
||||
|
||||
+168
-1
@@ -12,20 +12,25 @@ import { randomUUID } from 'node:crypto'
|
||||
import type { Database } from './connection.js'
|
||||
import { prepare } from './prepared.js'
|
||||
import {
|
||||
clusterInstanceId,
|
||||
toBusinessPlugin,
|
||||
toDomain,
|
||||
toDshHost,
|
||||
toDshInstance,
|
||||
toPublicUser,
|
||||
toSession,
|
||||
toUser,
|
||||
toWorkspace,
|
||||
type BusinessPlugin,
|
||||
type ClaimResult,
|
||||
type CredentialKey,
|
||||
type CredentialKeyMeta,
|
||||
type CredentialLandingRow,
|
||||
type CreateSessionInput,
|
||||
type CreateUserInput,
|
||||
type Domain,
|
||||
type DshHost,
|
||||
type DshHostStatus,
|
||||
type DshInstance,
|
||||
type DshInstanceRole,
|
||||
type DshInstanceStatus,
|
||||
@@ -33,6 +38,7 @@ import {
|
||||
type SessionRow,
|
||||
type SessionUser,
|
||||
type UpsertBusinessPluginInput,
|
||||
type UpsertDshHostInput,
|
||||
type UpsertDshInstanceInput,
|
||||
type User,
|
||||
type UserRole,
|
||||
@@ -43,7 +49,8 @@ const USER_COLS = 'id, username, pass_hash, role, home_dir, api_key_ref, created
|
||||
const DOMAIN_COLS = 'id, user_id, domain, verified, nginx_config, updated_at'
|
||||
const BUSINESS_PLUGIN_COLS = 'id, name, description, version, tgz_path, file_size, uploaded_by, created_at, updated_at'
|
||||
const INSTANCE_COLS =
|
||||
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch'
|
||||
'id, user_id, workspace_id, role, pid, port, status, started_at, last_exit, exit_code, last_error, folder, patch, '
|
||||
+ 'host_id, epoch, heartbeat_at, lease_until' // v7 集群化归属/租约(T08 S2)—— 漏了它们会让 hostId 恒为 null
|
||||
|
||||
export function createUser(db: Database, input: CreateUserInput, baseUid: number): User {
|
||||
const createdAt = Date.now()
|
||||
@@ -582,3 +589,163 @@ export function deleteBusinessPlugin(db: Database, id: string): boolean {
|
||||
const info = prepare(db, 'DELETE FROM business_plugins WHERE id = ?').run(id)
|
||||
return info.changes > 0
|
||||
}
|
||||
|
||||
// ── 集群化:worker 注册表 + 实例归属/租约(v7;T08 S2 / 设计 §3.1–§3.2)────────
|
||||
//
|
||||
// ⚠️ local 模式**不调用**这些函数(`LocalSpawner` 靠进程内 Map + 单机互斥),
|
||||
// 所以它们的存在不会改变现有单机行为。
|
||||
|
||||
const HOST_COLS = 'id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat'
|
||||
|
||||
/** 注册/更新一台 worker。join 幂等:同 id 重复执行 = 更新(并把它标回 `up`)。 */
|
||||
export function upsertDshHost(db: Database, input: UpsertDshHostInput): DshHost {
|
||||
prepare(db, `
|
||||
INSERT INTO dsh_hosts (id, endpoint, agent_token, capacity_mb, used_mb, status, last_heartbeat)
|
||||
VALUES (?, ?, ?, ?, 0, ?, NULL)
|
||||
ON CONFLICT(id) DO UPDATE SET
|
||||
endpoint = excluded.endpoint,
|
||||
agent_token = excluded.agent_token,
|
||||
capacity_mb = excluded.capacity_mb,
|
||||
status = excluded.status
|
||||
`).run(input.id, input.endpoint, input.agentToken, input.capacityMb, input.status ?? 'up')
|
||||
const row = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = ?`).get(input.id)
|
||||
return toDshHost(row as Record<string, unknown>)
|
||||
}
|
||||
|
||||
export function findDshHost(db: Database, id: string): DshHost | undefined {
|
||||
const row = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts WHERE id = ?`).get(id)
|
||||
return row ? toDshHost(row as Record<string, unknown>) : undefined
|
||||
}
|
||||
|
||||
export function listDshHosts(db: Database): DshHost[] {
|
||||
const rows = prepare(db, `SELECT ${HOST_COLS} FROM dsh_hosts ORDER BY id ASC`).all() as Array<
|
||||
Record<string, unknown>
|
||||
>
|
||||
return rows.map((row) => toDshHost(row))
|
||||
}
|
||||
|
||||
/** 心跳/状态上报(只更新显式给出的字段,避免 heartbeat 覆盖 status)。 */
|
||||
export function setDshHostStatus(
|
||||
db: Database,
|
||||
id: string,
|
||||
status: DshHostStatus,
|
||||
usedMb?: number,
|
||||
heartbeatAt?: number,
|
||||
): boolean {
|
||||
const info = prepare(db, `
|
||||
UPDATE dsh_hosts
|
||||
SET status = ?,
|
||||
used_mb = COALESCE(?, used_mb),
|
||||
last_heartbeat = COALESCE(?, last_heartbeat)
|
||||
WHERE id = ?
|
||||
`).run(status, usedMb ?? null, heartbeatAt ?? null, id)
|
||||
return info.changes > 0
|
||||
}
|
||||
|
||||
/**
|
||||
* **原子抢占**某用户 main 实例的归属(承重墙)。仅当"无人持有 **或** 租约已过期"才成功,
|
||||
* 成功时 `epoch` +1(fencing token)。失败时返回当前持有者与租约到期时刻。
|
||||
*/
|
||||
export function claimInstance(
|
||||
db: Database,
|
||||
userId: string,
|
||||
hostId: string,
|
||||
ttlMs: number,
|
||||
meta?: { folder?: string; patch?: string },
|
||||
): ClaimResult {
|
||||
const now = Date.now()
|
||||
const id = clusterInstanceId(userId)
|
||||
// 新用户没有 dsh_instances 行 ⇒ 先保证行存在(否则 UPDATE 影响 0 行被误判为"有人在管")。
|
||||
prepare(db, `
|
||||
INSERT INTO dsh_instances (id, user_id, role, status)
|
||||
VALUES (?, ?, 'main', 'starting')
|
||||
ON CONFLICT(id) DO NOTHING
|
||||
`).run(id, userId)
|
||||
// ⚠️ **必须把 folder/patch 一起落库**(2026-09-15 实测踩到):集群模式下实例行是这里建的,
|
||||
// 而 local 模式不写库 ⇒ 若这里不记,`folder` 永远是 NULL,**迁移时复现不了启动参数**
|
||||
// (表现为 `bwrap: Can't chdir to :` 空路径 ⇒ 崩溃循环)。用 COALESCE 保证不覆盖已有值。
|
||||
const info = prepare(db, `
|
||||
UPDATE dsh_instances
|
||||
SET host_id = ?, epoch = epoch + 1, heartbeat_at = ?, lease_until = ?,
|
||||
folder = COALESCE(?, folder), patch = COALESCE(?, patch)
|
||||
WHERE id = ? AND (host_id IS NULL OR lease_until < ?)
|
||||
`).run(hostId, now, now + ttlMs, meta?.folder ?? null, meta?.patch ?? null, id, now)
|
||||
const row = prepare(db, 'SELECT host_id, epoch, lease_until FROM dsh_instances WHERE id = ?').get(id) as
|
||||
| { host_id: string | null; epoch: number; lease_until: number }
|
||||
| undefined
|
||||
if (row === undefined) return { ok: false, holder: null, leaseUntil: 0 }
|
||||
return info.changes > 0
|
||||
? { ok: true, epoch: row.epoch, leaseUntil: row.lease_until }
|
||||
: { ok: false, holder: row.host_id, leaseUntil: row.lease_until }
|
||||
}
|
||||
|
||||
/** 续租。**必须带 epoch**:不匹配说明已被他人抢占 ⇒ 返回 false(fencing 生效)。 */
|
||||
export function renewInstanceLease(
|
||||
db: Database,
|
||||
userId: string,
|
||||
hostId: string,
|
||||
epoch: number,
|
||||
ttlMs: number,
|
||||
): boolean {
|
||||
const now = Date.now()
|
||||
const info = prepare(db, `
|
||||
UPDATE dsh_instances SET heartbeat_at = ?, lease_until = ?
|
||||
WHERE id = ? AND host_id = ? AND epoch = ?
|
||||
`).run(now, now + ttlMs, clusterInstanceId(userId), hostId, epoch)
|
||||
return info.changes > 0
|
||||
}
|
||||
|
||||
/**
|
||||
* 主动释放**租约**(停实例时)。带 epoch 校验,避免误清他人的归属。
|
||||
*
|
||||
* ⚠️ **只清 `lease_until`,保留 `host_id`**(2026-09-15 生产切换暴露):
|
||||
* `host_id` 的语义是「**这个用户的数据在哪台机器**」—— 用户的工作区在**本地盘**上,
|
||||
* 把归属一起清掉就等于**丢掉粘性锚点**,下次启动可能被调度到没有他数据的机器上(工作区看起来是空的)。
|
||||
* 「谁现在在托管」是**租约**(`lease_until`)的语义,所以释放只该清租约。
|
||||
*/
|
||||
export function releaseInstanceLease(db: Database, userId: string, hostId: string, epoch: number): boolean {
|
||||
const info = prepare(db, `
|
||||
UPDATE dsh_instances SET lease_until = 0
|
||||
WHERE id = ? AND host_id = ? AND epoch = ?
|
||||
`).run(clusterInstanceId(userId), hostId, epoch)
|
||||
return info.changes > 0
|
||||
}
|
||||
|
||||
/**
|
||||
* **钉住**某用户的归属(首次触达其工作区时用):只写 `host_id`,不动 epoch/租约。
|
||||
*
|
||||
* 为什么需要:新用户还没有归属,`selectHost` 会在**写文件那一步**与**launch 那一步**各自选一次,
|
||||
* 两次可能选到不同机器 ⇒ 「文件写到 A、实例起在 B」⇒ 实例看不到自己的文件(2026-09-15 实测)。
|
||||
* 首次触达就把归属钉住,后续(含 launch)都走粘性,两面必然一致。
|
||||
*/
|
||||
export function pinInstanceHost(db: Database, userId: string, hostId: string): void {
|
||||
const now = Date.now()
|
||||
const id = clusterInstanceId(userId)
|
||||
prepare(db, `
|
||||
INSERT INTO dsh_instances (id, user_id, role, status, host_id, epoch, heartbeat_at, lease_until)
|
||||
VALUES (?, ?, 'main', 'stopped', ?, 0, 0, 0)
|
||||
ON CONFLICT(id) DO NOTHING
|
||||
`).run(id, userId, hostId)
|
||||
prepare(db, `
|
||||
UPDATE dsh_instances SET host_id = ?
|
||||
WHERE id = ? AND (host_id IS NULL OR lease_until < ?)
|
||||
`).run(hostId, id, now)
|
||||
}
|
||||
|
||||
/** 租约过期但仍标着归属的 main 实例(供巡检/自愈;**不等于可以立即接管**,见 R9)。 */
|
||||
export function listExpiredInstanceLeases(db: Database, now: number): DshInstance[] {
|
||||
const rows = prepare(db, `
|
||||
SELECT ${INSTANCE_COLS} FROM dsh_instances
|
||||
WHERE role = 'main' AND host_id IS NOT NULL AND lease_until < ? AND status <> 'stopped'
|
||||
ORDER BY lease_until ASC
|
||||
`).all(now) as Array<Record<string, unknown>>
|
||||
return rows.map((row) => toDshInstance(row))
|
||||
}
|
||||
|
||||
/** 某 worker 上的全部实例 —— 对账**一次拿回整机**(替代逐用户查询)。 */
|
||||
export function listInstancesByHost(db: Database, hostId: string): DshInstance[] {
|
||||
const rows = prepare(db, `SELECT ${INSTANCE_COLS} FROM dsh_instances WHERE host_id = ? ORDER BY user_id ASC`).all(
|
||||
hostId,
|
||||
) as Array<Record<string, unknown>>
|
||||
return rows.map((row) => toDshInstance(row))
|
||||
}
|
||||
@@ -292,6 +292,59 @@ ALTER TABLE credential_vault ADD COLUMN models TEXT;
|
||||
ALTER TABLE users ADD COLUMN shared_model_enabled INTEGER NOT NULL DEFAULT 1;
|
||||
`
|
||||
|
||||
// v7: 集群化 —— worker 注册表 + 实例归属/租约(T08 S2;设计 §3.1/§3.2)。
|
||||
//
|
||||
// 为什么需要它:local 模式靠"进程内 Map + 单机"天然保证「一个用户只有一个活实例」;
|
||||
// 多机后这个保证必须落到 DB 的**原子 CAS** 上,否则两个 worker 会同时写同一个
|
||||
// `$DSH_HOME`(会话日志 append 冲突 ⇒ 数据损坏)。
|
||||
//
|
||||
// · `dsh_hosts` = worker 注册表:agent 地址、容量、水位、心跳时间。
|
||||
// · `dsh_instances.{host_id, epoch, heartbeat_at, lease_until}` = 归属与租约。
|
||||
// `epoch` 是 **fencing token**:抢占时 +1,旧持有者的写入据此被拒(防脑裂双写)。
|
||||
//
|
||||
// 抢占语义(两方言同款,见 `repo.ts` 的 claimInstance / `pg.ts` 同名方法):
|
||||
// `INSERT … ON CONFLICT(id) DO UPDATE SET … WHERE host_id IS NULL OR lease_until < now`
|
||||
// —— 冲突时仅在"无人持有或租约过期"才更新;否则**不动行也不报错**,
|
||||
// 调用方以「受影响行数 0」判定"有人在管"。
|
||||
// ⚠️ 时间戳一律 **epoch 毫秒 BIGINT**(与全库一致,勿用 timestamptz)。
|
||||
const SQLITE_V7 = `
|
||||
CREATE TABLE IF NOT EXISTS dsh_hosts (
|
||||
id TEXT PRIMARY KEY,
|
||||
endpoint TEXT NOT NULL,
|
||||
agent_token TEXT NOT NULL,
|
||||
capacity_mb INTEGER NOT NULL DEFAULT 0,
|
||||
used_mb INTEGER NOT NULL DEFAULT 0,
|
||||
status TEXT NOT NULL DEFAULT 'up'
|
||||
CHECK (status IN ('up','draining','down')),
|
||||
last_heartbeat INTEGER
|
||||
);
|
||||
ALTER TABLE dsh_instances ADD COLUMN host_id TEXT;
|
||||
ALTER TABLE dsh_instances ADD COLUMN epoch INTEGER NOT NULL DEFAULT 0;
|
||||
ALTER TABLE dsh_instances ADD COLUMN heartbeat_at INTEGER NOT NULL DEFAULT 0;
|
||||
ALTER TABLE dsh_instances ADD COLUMN lease_until INTEGER NOT NULL DEFAULT 0;
|
||||
CREATE INDEX IF NOT EXISTS idx_dsh_instances_host ON dsh_instances (host_id);
|
||||
CREATE INDEX IF NOT EXISTS idx_dsh_instances_lease ON dsh_instances (lease_until);
|
||||
`
|
||||
|
||||
const PG_V7 = `
|
||||
CREATE TABLE IF NOT EXISTS dsh_hosts (
|
||||
id TEXT PRIMARY KEY,
|
||||
endpoint TEXT NOT NULL,
|
||||
agent_token TEXT NOT NULL,
|
||||
capacity_mb BIGINT NOT NULL DEFAULT 0,
|
||||
used_mb BIGINT NOT NULL DEFAULT 0,
|
||||
status TEXT NOT NULL DEFAULT 'up'
|
||||
CHECK (status IN ('up','draining','down')),
|
||||
last_heartbeat BIGINT
|
||||
);
|
||||
ALTER TABLE dsh_instances ADD COLUMN host_id TEXT;
|
||||
ALTER TABLE dsh_instances ADD COLUMN epoch BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE dsh_instances ADD COLUMN heartbeat_at BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE dsh_instances ADD COLUMN lease_until BIGINT NOT NULL DEFAULT 0;
|
||||
CREATE INDEX IF NOT EXISTS idx_dsh_instances_host ON dsh_instances (host_id);
|
||||
CREATE INDEX IF NOT EXISTS idx_dsh_instances_lease ON dsh_instances (lease_until);
|
||||
`
|
||||
|
||||
interface Migration {
|
||||
version: number
|
||||
name: string
|
||||
@@ -306,6 +359,7 @@ const MIGRATIONS: readonly Migration[] = [
|
||||
{ version: 4, name: 'instance desired state', sqlite: SQLITE_V4, pg: PG_V4 },
|
||||
{ version: 5, name: 'business plugin candidate pool', sqlite: SQLITE_V5, pg: PG_V5 },
|
||||
{ version: 6, name: 'user model providers', sqlite: SQLITE_V6, pg: PG_V6 },
|
||||
{ version: 7, name: 'cluster host registry + instance lease', sqlite: SQLITE_V7, pg: PG_V7 },
|
||||
]
|
||||
|
||||
/** Apply unapplied SQLite migrations inside a single transaction. */
|
||||
|
||||
@@ -57,18 +57,32 @@ import {
|
||||
setUserRole as setUserRoleSync,
|
||||
setUserUid as setUserUidSync,
|
||||
toggleCredentialKey as toggleCredentialKeySync,
|
||||
// 集群化(v7;T08 S2)
|
||||
claimInstance as claimInstanceSync,
|
||||
findDshHost as findDshHostSync,
|
||||
listDshHosts as listDshHostsSync,
|
||||
listExpiredInstanceLeases as listExpiredInstanceLeasesSync,
|
||||
listInstancesByHost as listInstancesByHostSync,
|
||||
pinInstanceHost as pinInstanceHostSync,
|
||||
releaseInstanceLease as releaseInstanceLeaseSync,
|
||||
renewInstanceLease as renewInstanceLeaseSync,
|
||||
setDshHostStatus as setDshHostStatusSync,
|
||||
upsertDshHost as upsertDshHostSync,
|
||||
upsertBusinessPlugin as upsertBusinessPluginSync,
|
||||
upsertDomain as upsertDomainSync,
|
||||
upsertInstance as upsertInstanceSync,
|
||||
} from './repo.js'
|
||||
import type {
|
||||
BusinessPlugin,
|
||||
ClaimResult,
|
||||
CredentialKey,
|
||||
CredentialKeyMeta,
|
||||
CredentialLandingRow,
|
||||
CreateSessionInput,
|
||||
CreateUserInput,
|
||||
Domain,
|
||||
DshHost,
|
||||
DshHostStatus,
|
||||
DshInstance,
|
||||
DshInstanceRole,
|
||||
DshInstanceStatus,
|
||||
@@ -76,6 +90,7 @@ import type {
|
||||
SessionRow,
|
||||
SessionUser,
|
||||
UpsertBusinessPluginInput,
|
||||
UpsertDshHostInput,
|
||||
UpsertDshInstanceInput,
|
||||
User,
|
||||
UserRole,
|
||||
@@ -321,6 +336,64 @@ export class SqliteAdapter implements DbAdapter {
|
||||
deleteUserInstancesSync(this.db, userId)
|
||||
}
|
||||
|
||||
// ── 集群化:worker 注册表 + 归属/租约(v7;T08 S2)────────────────────────
|
||||
// local 模式不会走到这些方法(`LocalSpawner` 不写库),它们只是让
|
||||
// **SQLite 侧与 PG 侧行为一致** —— 测试与单机试跑都需要。
|
||||
|
||||
async upsertDshHost(input: UpsertDshHostInput): Promise<DshHost> {
|
||||
try {
|
||||
return upsertDshHostSync(this.db, input)
|
||||
} catch (e) {
|
||||
mapSqliteError(e)
|
||||
}
|
||||
}
|
||||
|
||||
async findDshHost(id: string): Promise<DshHost | undefined> {
|
||||
return findDshHostSync(this.db, id)
|
||||
}
|
||||
|
||||
async listDshHosts(): Promise<DshHost[]> {
|
||||
return listDshHostsSync(this.db)
|
||||
}
|
||||
|
||||
async setDshHostStatus(
|
||||
id: string,
|
||||
status: DshHostStatus,
|
||||
usedMb?: number,
|
||||
heartbeatAt?: number,
|
||||
): Promise<boolean> {
|
||||
return setDshHostStatusSync(this.db, id, status, usedMb, heartbeatAt)
|
||||
}
|
||||
|
||||
async claimInstance(
|
||||
userId: string,
|
||||
hostId: string,
|
||||
ttlMs: number,
|
||||
meta?: { folder?: string; patch?: string },
|
||||
): Promise<ClaimResult> {
|
||||
return claimInstanceSync(this.db, userId, hostId, ttlMs, meta)
|
||||
}
|
||||
|
||||
async renewInstanceLease(userId: string, hostId: string, epoch: number, ttlMs: number): Promise<boolean> {
|
||||
return renewInstanceLeaseSync(this.db, userId, hostId, epoch, ttlMs)
|
||||
}
|
||||
|
||||
async releaseInstanceLease(userId: string, hostId: string, epoch: number): Promise<boolean> {
|
||||
return releaseInstanceLeaseSync(this.db, userId, hostId, epoch)
|
||||
}
|
||||
|
||||
async pinInstanceHost(userId: string, hostId: string): Promise<void> {
|
||||
pinInstanceHostSync(this.db, userId, hostId)
|
||||
}
|
||||
|
||||
async listExpiredInstanceLeases(now: number): Promise<DshInstance[]> {
|
||||
return listExpiredInstanceLeasesSync(this.db, now)
|
||||
}
|
||||
|
||||
async listInstancesByHost(hostId: string): Promise<DshInstance[]> {
|
||||
return listInstancesByHostSync(this.db, hostId)
|
||||
}
|
||||
|
||||
async close(): Promise<void> {
|
||||
this.db.close()
|
||||
}
|
||||
|
||||
@@ -90,6 +90,15 @@ export interface DshInstance {
|
||||
folder: string | null
|
||||
/** Rendered Cordis patch content (not a path — the control plane holds no user volume). */
|
||||
patch: string | null
|
||||
// ── 集群化归属与租约(v7;T08 S2 / 设计 §3.1)—— local 模式下恒为 null/0 ──
|
||||
/** 托管该实例的 worker(`dsh_hosts.id`);null = 未被任何 worker 认领。 */
|
||||
hostId: string | null
|
||||
/** **fencing token**:每次抢占 +1;旧持有者的写入据此被拒(防脑裂双写)。 */
|
||||
epoch: number
|
||||
/** 最近一次心跳(epoch 毫秒)。 */
|
||||
heartbeatAt: number
|
||||
/** 租约到期时刻(epoch 毫秒);早于 now 即可被他人抢占。 */
|
||||
leaseUntil: number
|
||||
}
|
||||
|
||||
/** A named per-user credential key (secret never exposed). */
|
||||
@@ -262,6 +271,10 @@ export function toDshInstance(row: Record<string, unknown>): DshInstance {
|
||||
lastError: (row.last_error as string | null) ?? null,
|
||||
folder: (row.folder as string | null) ?? null,
|
||||
patch: (row.patch as string | null) ?? null,
|
||||
hostId: (row.host_id as string | null) ?? null,
|
||||
epoch: (row.epoch as number | null) ?? 0,
|
||||
heartbeatAt: (row.heartbeat_at as number | null) ?? 0,
|
||||
leaseUntil: (row.lease_until as number | null) ?? 0,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -289,3 +302,64 @@ export function toBusinessPlugin(row: Record<string, unknown>): BusinessPlugin {
|
||||
updatedAt: row.updated_at as number,
|
||||
}
|
||||
}
|
||||
|
||||
// ── 集群化:worker 注册表与租约结果(v7;T08 S2 / 设计 §3.1–§3.2)──────────────
|
||||
|
||||
/** Worker 健康状态(`dsh_hosts.status` 的 CHECK 镜像)。 */
|
||||
export type DshHostStatus = 'up' | 'draining' | 'down'
|
||||
|
||||
/** 一台承载用户实例的 worker(= 设计里的 Worker 节点)。 */
|
||||
export interface DshHost {
|
||||
id: string
|
||||
/** agent 的内网地址,如 `10.0.1.11:9000`。 */
|
||||
endpoint: string
|
||||
/** 内部 HMAC 密钥(**只应存在于 DB 与 Manager 内存**,绝不经 API 返回)。 */
|
||||
agentToken: string
|
||||
/** 该机可用内存预算(MB);0 = 不承载实例(只做门户/控制)。 */
|
||||
capacityMb: number
|
||||
/** 由心跳上报的已用内存(MB)。 */
|
||||
usedMb: number
|
||||
status: DshHostStatus
|
||||
/** 最近心跳(epoch 毫秒);null = 从未上报。 */
|
||||
lastHeartbeat: number | null
|
||||
}
|
||||
|
||||
/** Upsert payload for `dsh_hosts`(join 脚本/管理面用)。 */
|
||||
export interface UpsertDshHostInput {
|
||||
id: string
|
||||
endpoint: string
|
||||
agentToken: string
|
||||
capacityMb: number
|
||||
status?: DshHostStatus
|
||||
}
|
||||
|
||||
/**
|
||||
* 抢占结果。`ok:false` 时带回**当前持有者**与租约到期时刻,便于调用方决定
|
||||
* "退让"还是"报告异常"(**不要据此接管** —— 见项目红线 R9)。
|
||||
*/
|
||||
export type ClaimResult =
|
||||
| { ok: true; epoch: number; leaseUntil: number }
|
||||
| { ok: false; holder: string | null; leaseUntil: number }
|
||||
|
||||
/**
|
||||
* 集群模式下 main 实例行的**确定性 id**。
|
||||
*
|
||||
* 为什么需要确定性:租约是以 **(user, role='main')** 为单位的,`dsh_instances.id` 只是载体;
|
||||
* 若每次 spawn 用随机 id,抢占时会插出多行 ⇒ 归属判断失效。local 模式仍用随机 id
|
||||
* (它不写库),集群路径一律走这里。
|
||||
*/
|
||||
export function clusterInstanceId(userId: string): string {
|
||||
return `dsh-${userId}`
|
||||
}
|
||||
|
||||
export function toDshHost(row: Record<string, unknown>): DshHost {
|
||||
return {
|
||||
id: row.id as string,
|
||||
endpoint: row.endpoint as string,
|
||||
agentToken: row.agent_token as string,
|
||||
capacityMb: (row.capacity_mb as number | null) ?? 0,
|
||||
usedMb: (row.used_mb as number | null) ?? 0,
|
||||
status: row.status as DshHostStatus,
|
||||
lastHeartbeat: (row.last_heartbeat as number | null) ?? null,
|
||||
}
|
||||
}
|
||||
+21
-3
@@ -7,13 +7,31 @@
|
||||
|
||||
import type { ServerConfig } from '../config.js'
|
||||
import { LocalUserFs } from './local-user-fs.js'
|
||||
import { RemoteUserFs } from './remote-user-fs.js'
|
||||
import type { UserFs } from './user-fs.js'
|
||||
import { userRoot } from './workspace.js'
|
||||
|
||||
/** cluster 模式下的按用户路由(由 `server.ts` 注入;见 RemoteUserFsOptions 的说明)。 */
|
||||
export interface ClusterFsRouting {
|
||||
hostIdFor?: (userId: string) => Promise<string | undefined>
|
||||
agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the configured per-user filesystem. The single-machine backend
|
||||
* touches the users volume in-process.
|
||||
* Build the configured per-user filesystem.
|
||||
* - `local` :控制面**在本进程内**直接碰用户卷(单机形态)。
|
||||
* - `cluster` :用户卷在 **worker** 上 ⇒ 走 agent 的 `/fs/*`(T08 S5)——
|
||||
* 远端实现复用同一份路径安全逻辑,`resolvePath` 按 **worker 的 dataRoot** 做路径数学。
|
||||
*/
|
||||
export function createUserFs(config: ServerConfig): UserFs {
|
||||
export function createUserFs(config: ServerConfig, routing: ClusterFsRouting = {}): UserFs {
|
||||
if (config.deployMode === 'cluster') {
|
||||
return new RemoteUserFs({
|
||||
agentUrl: config.clusterAgentUrl,
|
||||
token: config.clusterAgentToken,
|
||||
workerDataRoot: config.clusterWorkerDataRoot === '' ? config.dataRoot : config.clusterWorkerDataRoot,
|
||||
hostIdFor: routing.hostIdFor,
|
||||
agentFor: routing.agentFor,
|
||||
})
|
||||
}
|
||||
return new LocalUserFs((userId) => userRoot(config.dataRoot, userId))
|
||||
}
|
||||
@@ -0,0 +1,180 @@
|
||||
/**
|
||||
* `UserFs` 的**远端实现**(T08 S5)。
|
||||
*
|
||||
* 为什么需要它:门户的「我的文件」(`/api/desktop/tree`、`/api/fs/*`) 只依赖 `UserFs` seam;
|
||||
* 多机后用户卷在 **worker** 上,Manager 的本地读会落空。做法不是重新实现一套路径语义,
|
||||
* 而是把 worker 上**同一个 `LocalUserFs`** 经 agent 的 `/fs/*` 暴露出来 —— 路径安全
|
||||
* (`resolveWithinRoot` / `safeFilename` / `PathEscapeError`)**继续复用同一份代码**,
|
||||
* 所以"本地能过的路径,远端行为一致"是结构性保证,不是靠测试碰运气。
|
||||
*
|
||||
* `resolvePath` 是**纯路径数学**(同步接口),按 **worker 的 dataRoot** 计算 ——
|
||||
* 这正是"实例眼里的路径"。因此多机部署有一条**基线约定**:
|
||||
* **所有 worker 的 dataRoot 必须是同一个绝对路径**(同镜像即可满足,见设计 §14.3 机器基线)。
|
||||
* 不一致时 `buildServer` 会在启动时把差异**报出来**(见 `server.ts` 的 probe)。
|
||||
*
|
||||
* @module dshs/fs/remote-user-fs
|
||||
*/
|
||||
import { AGENT_TOKEN_HEADER } from '../worker/agent.js'
|
||||
import { PathEscapeError, resolveWithinRoot } from '../web/middleware/fs-guard.js'
|
||||
import type { PluginInfo } from './plugins.js'
|
||||
import { UserFsError, isUserFsErrorCode, type UserFs } from './user-fs.js'
|
||||
import type { FsEntry } from './workspace.js'
|
||||
import { userRoot, workspaceRoot } from './workspace.js'
|
||||
|
||||
export interface RemoteUserFsOptions {
|
||||
/** **默认/回退** worker agent 基址(未提供 hostIdFor 或查不到归属时用它)。 */
|
||||
agentUrl: string
|
||||
/** 与默认 agent 约定的共享密钥。 */
|
||||
token: string
|
||||
/**
|
||||
* **按用户归属路由**(2026-09-15 生产切换暴露的缺口)。
|
||||
*
|
||||
* 为什么必须有:用户工作区在**那台 worker 的本地盘**上;文件面若固定打一台 agent,
|
||||
* 就会出现「实例跑在 A、而 mkdir/上传写到 B」⇒ 实例看不到自己的文件、甚至 cwd 不存在而崩。
|
||||
* 传 `hostIdFor`(查 `dsh_instances.host_id`)即可让每次文件操作落到**该用户所在的机器**。
|
||||
*/
|
||||
hostIdFor?: (userId: string) => Promise<string | undefined>
|
||||
/** hostId → 接入信息(与 RemoteSpawner 用**同一份**目录,避免两套漂移)。 */
|
||||
agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
|
||||
/** **worker 上**的 dataRoot(必须与该 worker 一致,用于 `resolvePath` 的路径数学)。 */
|
||||
workerDataRoot: string
|
||||
/** 单次请求超时(ms)。文件可能较大,默认 30 s。 */
|
||||
timeoutMs?: number
|
||||
fetchImpl?: typeof fetch
|
||||
}
|
||||
|
||||
export class RemoteUserFs implements UserFs {
|
||||
private readonly base: string
|
||||
private readonly token: string
|
||||
/** **worker 上**的 dataRoot —— 公开只读,供 `server.ts` 启动时做基线一致性探测。 */
|
||||
readonly workerDataRoot: string
|
||||
private readonly timeoutMs: number
|
||||
private readonly doFetch: typeof fetch
|
||||
private readonly hostIdFor?: (userId: string) => Promise<string | undefined>
|
||||
private readonly agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
|
||||
|
||||
constructor(options: RemoteUserFsOptions) {
|
||||
this.base = options.agentUrl.replace(/\/$/, '')
|
||||
this.token = options.token
|
||||
this.hostIdFor = options.hostIdFor
|
||||
this.agentFor = options.agentFor
|
||||
this.workerDataRoot = options.workerDataRoot
|
||||
this.timeoutMs = options.timeoutMs ?? 30_000
|
||||
this.doFetch = options.fetchImpl ?? fetch
|
||||
}
|
||||
|
||||
/** 解析该用户文件操作应打的那台 agent(查不到归属就用默认)。 */
|
||||
private async target(userId: string): Promise<{ base: string; token: string }> {
|
||||
if (this.hostIdFor === undefined) return { base: this.base, token: this.token }
|
||||
try {
|
||||
const hostId = await this.hostIdFor(userId)
|
||||
if (hostId !== undefined && hostId !== null && this.agentFor !== undefined) {
|
||||
const agent = this.agentFor(hostId)
|
||||
if (agent !== undefined) return { base: agent.agentUrl.replace(/\/$/, ''), token: agent.token }
|
||||
}
|
||||
} catch {
|
||||
// 查库失败 ⇒ 落回默认 host(宁可"可能读错机",也不要整个文件面 500)
|
||||
}
|
||||
return { base: this.base, token: this.token }
|
||||
}
|
||||
|
||||
/** 统一的 POST:把 agent 的 `{error: code}` 还原成 `UserFsError`(路由按 code 回前端)。 */
|
||||
private async post<T>(path: string, body: Record<string, unknown>): Promise<T> {
|
||||
const userId = typeof body.userId === 'string' ? body.userId : ''
|
||||
const t = await this.target(userId)
|
||||
const res = await this.doFetch(`${t.base}${path}`, {
|
||||
method: 'POST',
|
||||
headers: { [AGENT_TOKEN_HEADER]: t.token, 'content-type': 'application/json' },
|
||||
body: JSON.stringify(body),
|
||||
signal: AbortSignal.timeout(this.timeoutMs),
|
||||
})
|
||||
const text = await res.text()
|
||||
if (res.ok) return (text === '' ? undefined : JSON.parse(text)) as T
|
||||
let code: unknown
|
||||
try {
|
||||
code = (JSON.parse(text) as { error?: unknown }).error
|
||||
} catch {
|
||||
code = undefined
|
||||
}
|
||||
if (typeof code === 'string' && isUserFsErrorCode(code)) throw new UserFsError(code)
|
||||
throw new Error(`agent POST ${path} → ${res.status}: ${text.slice(0, 200)}`)
|
||||
}
|
||||
|
||||
async initUserRoot(userId: string, uid?: number): Promise<void> {
|
||||
// 注意:uid 也要带过去 —— 本地实现会 chown 用户根,远端由 worker 执行同一动作。
|
||||
await this.post('/fs/init', uid === undefined ? { userId } : { userId, uid })
|
||||
}
|
||||
|
||||
/**
|
||||
* **纯路径数学**,按 worker 的 dataRoot 算(= 实例眼里的绝对路径)。
|
||||
* 刻意**不**像本地实现那样 `ensureDir` —— Manager 不该在**自己**的盘上造目录。
|
||||
*/
|
||||
resolvePath(userId: string, relPath: string): string {
|
||||
const ws = workspaceRoot(userRoot(this.workerDataRoot, userId))
|
||||
try {
|
||||
return resolveWithinRoot(ws, relPath)
|
||||
} catch (err) {
|
||||
if (err instanceof PathEscapeError) throw new UserFsError('bad_path')
|
||||
throw err
|
||||
}
|
||||
}
|
||||
|
||||
async listDir(userId: string, relPath: string): Promise<FsEntry[]> {
|
||||
return this.post<FsEntry[]>('/fs/list', { userId, relPath })
|
||||
}
|
||||
|
||||
async mkdir(userId: string, relPath: string): Promise<void> {
|
||||
await this.post('/fs/mkdir', { userId, relPath })
|
||||
}
|
||||
|
||||
async createEntry(userId: string, relPath: string, name: string, type: 'file' | 'dir'): Promise<string> {
|
||||
const out = await this.post<{ name: string }>('/fs/create', { userId, relPath, name, type })
|
||||
return out.name
|
||||
}
|
||||
|
||||
async upload(userId: string, relPath: string, name: string, data: Buffer): Promise<string> {
|
||||
const out = await this.post<{ name: string }>('/fs/upload', {
|
||||
userId,
|
||||
relPath,
|
||||
name,
|
||||
dataBase64: data.toString('base64'),
|
||||
})
|
||||
return out.name
|
||||
}
|
||||
|
||||
async isDirectory(userId: string, relPath: string): Promise<boolean> {
|
||||
const out = await this.post<{ isDirectory: boolean }>('/fs/isdir', { userId, relPath })
|
||||
return out.isDirectory
|
||||
}
|
||||
|
||||
async readFile(userId: string, relPath: string, maxBytes?: number): Promise<{ name: string; data: Buffer }> {
|
||||
const out = await this.post<{ name: string; dataBase64: string }>(
|
||||
'/fs/read',
|
||||
maxBytes === undefined ? { userId, relPath } : { userId, relPath, maxBytes },
|
||||
)
|
||||
return { name: out.name, data: Buffer.from(out.dataBase64, 'base64') }
|
||||
}
|
||||
|
||||
async listInstalledPlugins(userId: string): Promise<PluginInfo[]> {
|
||||
return this.post<PluginInfo[]>('/fs/plugins', { userId })
|
||||
}
|
||||
|
||||
async writeHandoff(userId: string, content: string): Promise<void> {
|
||||
await this.post('/fs/handoff', { userId, content })
|
||||
}
|
||||
|
||||
/** 探测 worker 的 dataRoot(用于启动时的基线一致性检查,见 `server.ts`)。 */
|
||||
async probeWorkerRoot(): Promise<string | undefined> {
|
||||
try {
|
||||
const res = await this.doFetch(`${this.base}/fs/root`, {
|
||||
headers: { [AGENT_TOKEN_HEADER]: this.token },
|
||||
signal: AbortSignal.timeout(5_000),
|
||||
})
|
||||
if (!res.ok) return undefined
|
||||
const body = (await res.json()) as { dataRoot?: string }
|
||||
return body.dataRoot
|
||||
} catch {
|
||||
return undefined
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,169 @@
|
||||
/**
|
||||
* 实例归属租约(T08 S2;设计 §3.1–§3.2)。
|
||||
*
|
||||
* **为什么必须有它**:local 模式靠"进程内 Map + 单机"天然保证「一个用户同时只有一个活实例」;
|
||||
* 多机后这个保证只能落到 **DB 的原子 CAS** 上 —— 否则两个 worker 会同时写同一个 `$DSH_HOME`
|
||||
* (会话日志 append 冲突 ⇒ **数据损坏**,本库最贵的一类事故)。
|
||||
*
|
||||
* 三条不变量(都不许省):
|
||||
* 1. **单写者**:抢占必须原子(`claimInstance` 的 `UPDATE … WHERE 无人持有 OR 租约过期`),
|
||||
* 调用方以"是否真正更新到行"判定成败,**不许"先读后写"**。
|
||||
* 2. **TTL > 2 × 续租间隔**:留足抖动余量;否则自身网络一抖就会误判自己失权
|
||||
* (或更糟 —— 误判别人已死)。构造时**硬校验**,fail-loud。
|
||||
* 3. **fencing**:每次操作带 `epoch`;不匹配 = 已被他人抢占 ⇒ 调用方必须**自杀**
|
||||
* (self-fencing,如停掉自己那个实例),而不是继续写。
|
||||
*
|
||||
* ⚠️ 与 **R9** 的关系:本模块只提供"判定与递增 epoch"的机械能力,
|
||||
* **不提供**"判定对方已死 → 接管"的自动化。没有心跳判据时单方面接管是被明令禁止的;
|
||||
* 因此 `expiredAll()` 只用于**巡检/报告/人工确认后的动作**,绝不自动接管。
|
||||
*
|
||||
* @module dshs/supervisor/lease
|
||||
*/
|
||||
import type { DbAdapter } from '../db/adapter.js'
|
||||
import type { ClaimResult, DshInstance } from '../db/types.js'
|
||||
|
||||
/** 默认租约存活 30 s(设计 §3.2 的建议时序)。 */
|
||||
export const DEFAULT_LEASE_TTL_MS = 30_000
|
||||
/** 默认续租间隔 10 s(不变量:ttl > 2 × renew)。 */
|
||||
export const DEFAULT_LEASE_RENEW_MS = 10_000
|
||||
|
||||
export interface LeaseOptions {
|
||||
/** 租约存活时长(ms)。不变量:必须 **> 2 × renewMs**。 */
|
||||
ttlMs?: number
|
||||
/** 续租间隔(ms)。 */
|
||||
renewMs?: number
|
||||
/** 注入时钟 —— 仅用于本类自己的判定(**SQL 里的 now 仍是 `Date.now()`**)。 */
|
||||
now?: () => number
|
||||
}
|
||||
|
||||
/**
|
||||
* 某实例当前是否由「我」合法持有(fencing 判据)。
|
||||
*
|
||||
* 用在两处:① 收到心跳/指令前自检"我还是不是持有者"② 老 worker 复活后判断
|
||||
* 自己**是否已被接管** ⇒ 是则自杀(防双写)。
|
||||
*/
|
||||
export function stillHolder(
|
||||
instance: DshInstance | undefined,
|
||||
hostId: string,
|
||||
epoch: number,
|
||||
): boolean {
|
||||
return instance !== undefined && instance.hostId === hostId && instance.epoch === epoch
|
||||
}
|
||||
|
||||
/** 一个用户实例的租约句柄(holding = 我持有 + 我的 epoch)。 */
|
||||
export class InstanceLease {
|
||||
readonly hostId: string
|
||||
private readonly db: DbAdapter
|
||||
private readonly ttl: number
|
||||
private readonly renewInterval: number
|
||||
private readonly now: () => number
|
||||
/**
|
||||
* userId → **我认领到的那把租约**(epoch + 落在哪个 host)。
|
||||
*
|
||||
* 为什么必须记住 hostId:多 worker 后 `renewInstanceLease(userId, hostId, epoch)` 要能在
|
||||
* **正确的那个 host** 上校验;只记 epoch 会在多机下续错对象(2026-09-15 T08 S6)。
|
||||
*/
|
||||
private readonly held = new Map<string, { epoch: number; hostId: string }>()
|
||||
|
||||
constructor(db: DbAdapter, hostId: string, options: LeaseOptions = {}) {
|
||||
this.db = db
|
||||
this.hostId = hostId
|
||||
this.ttl = options.ttlMs ?? DEFAULT_LEASE_TTL_MS
|
||||
this.renewInterval = options.renewMs ?? DEFAULT_LEASE_RENEW_MS
|
||||
this.now = options.now ?? Date.now
|
||||
// 不变量:TTL 必须显著大于续租间隔,否则单次网络抖动就会造成"自己失权"或"误判他人已死"。
|
||||
if (this.ttl <= 2 * this.renewInterval) {
|
||||
throw new Error(
|
||||
`lease: ttlMs(${this.ttl}) must be > 2 × renewMs(${this.renewInterval}) —— ` +
|
||||
'否则时钟/网络抖动会破坏单写者保证(设计 §3.2 不变量 2)',
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
get ttlMs(): number {
|
||||
return this.ttl
|
||||
}
|
||||
|
||||
get renewMs(): number {
|
||||
return this.renewInterval
|
||||
}
|
||||
|
||||
/** 我当前认领的实例(userId → {epoch, hostId})。 */
|
||||
holdings(): ReadonlyMap<string, { epoch: number; hostId: string }> {
|
||||
return this.held
|
||||
}
|
||||
|
||||
/**
|
||||
* 抢占某用户 main 实例的归属。
|
||||
*
|
||||
* `ok:false` = 有人在管 ⇒ **退让**:不要接管、不要重试到死,交给上层决定
|
||||
* (拉长等待 / 报告管理员)。成功时记住 epoch 供续租与 fencing 使用。
|
||||
*/
|
||||
async acquire(
|
||||
userId: string,
|
||||
hostId?: string,
|
||||
meta?: { folder?: string; patch?: string },
|
||||
): Promise<ClaimResult> {
|
||||
const target = hostId ?? this.hostId
|
||||
// meta(folder/patch)随认领一起落库 ⇒ 迁移才能复现启动参数
|
||||
const res = await this.db.claimInstance(userId, target, this.ttl, meta)
|
||||
if (res.ok) this.held.set(userId, { epoch: res.epoch, hostId: target })
|
||||
return res
|
||||
}
|
||||
|
||||
/**
|
||||
* 续租。返回 false = **我已失权**(被他人以更高 epoch 抢占,或行被删)⇒ 调用方
|
||||
* 必须 self-fence(停掉自己那个实例),并清掉本地记录。
|
||||
*/
|
||||
async renew(userId: string): Promise<boolean> {
|
||||
const held = this.held.get(userId)
|
||||
if (held === undefined) return false
|
||||
const ok = await this.db.renewInstanceLease(userId, held.hostId, held.epoch, this.ttl)
|
||||
if (!ok) this.held.delete(userId)
|
||||
return ok
|
||||
}
|
||||
|
||||
/** 批量续租(心跳 tick 用)。返回失权的 userId 列表(调用方据此 self-fence)。 */
|
||||
async renewAll(): Promise<string[]> {
|
||||
const lost: string[] = []
|
||||
for (const userId of [...this.held.keys()]) {
|
||||
if (!(await this.renew(userId))) lost.push(userId)
|
||||
}
|
||||
return lost
|
||||
}
|
||||
|
||||
/** 主动释放(停实例时)。成功后不再持有该用户。 */
|
||||
async release(userId: string): Promise<boolean> {
|
||||
const held = this.held.get(userId)
|
||||
if (held === undefined) return false
|
||||
const ok = await this.db.releaseInstanceLease(userId, held.hostId, held.epoch)
|
||||
if (ok) this.held.delete(userId)
|
||||
return ok
|
||||
}
|
||||
|
||||
/** 重新读取 DB 里的真实归属,校正本地记录(对账用)。 */
|
||||
async refresh(userId: string): Promise<DshInstance | undefined> {
|
||||
const inst = await this.db.findUserInstance(userId, 'main')
|
||||
const held = this.held.get(userId)
|
||||
if (!stillHolder(inst, held?.hostId ?? this.hostId, held?.epoch ?? -1)) this.held.delete(userId)
|
||||
return inst
|
||||
}
|
||||
|
||||
/** **我名下**租约已过期的实例(供巡检;**不等于可以接管**,见模块头与 R9)。 */
|
||||
async expiredHere(): Promise<DshInstance[]> {
|
||||
const all = await this.db.listExpiredInstanceLeases(this.now())
|
||||
const mine = new Set([...this.held.values()].map((h) => h.hostId))
|
||||
mine.add(this.hostId)
|
||||
return all.filter((inst) => inst.hostId !== null && mine.has(inst.hostId))
|
||||
}
|
||||
|
||||
/** 全集群租约已过期的实例(管理面/巡检用;**不自动接管**)。 */
|
||||
async expiredAll(): Promise<DshInstance[]> {
|
||||
return this.db.listExpiredInstanceLeases(this.now())
|
||||
}
|
||||
|
||||
/** 对账:**一次拿回本机全部实例**(替代逐用户查询,设计 §11.6)。 */
|
||||
async mine(): Promise<DshInstance[]> {
|
||||
return this.db.listInstancesByHost(this.hostId)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,284 @@
|
||||
/**
|
||||
* 给任意 `Spawner` 套上**归属租约**(T08 S4;设计 §1.2/§3.2/§11.5)。
|
||||
*
|
||||
* 它补上集群模式下 Manager 侧最关键的一环:**"能不能拉起"必须先问过归属**。
|
||||
* 本地模式靠"进程内 Map + 单机"天然保证单写者;多机后这个保证只能落在 DB 的原子 CAS 上
|
||||
* —— 两个 Manager 各持一把租约、抢同一个用户,就是**双写同一个 home** = 数据损坏。
|
||||
*
|
||||
* 三个动作:
|
||||
* 1. `launch` 前 **claim**:拿不到就抛 `LeaseBusyError`(**退让**,不是接管 —— 见 R9);
|
||||
* 2. 心跳里 **renewAll**:续租;同时把"我已失权"的实例用 **`/fence`** 通知 worker 停掉
|
||||
* (self-fencing 的 Manager 侧对齐,设计 §11.5);
|
||||
* 3. `stop` 时 **release**:归属交还,别人立刻可以接管(不用等 TTL)。
|
||||
*
|
||||
* ⚠️ 归属只由 Manager 写(设计 §1.3 数据分层的判据 1)。worker 侧只被通知。
|
||||
*
|
||||
* @module dshs/supervisor/leased-spawner
|
||||
*/
|
||||
import { AGENT_TOKEN_HEADER } from '../worker/agent.js'
|
||||
import type { DbAdapter } from '../db/adapter.js'
|
||||
import type { Endpoint, Instance, Spawner, UserStatus } from './spawner.js'
|
||||
import { InstanceLease, type LeaseOptions } from './lease.js'
|
||||
|
||||
/** 归属被别人持有时抛出 —— 调用方应**退让**(等待/报告),**不得接管**(R9)。 */
|
||||
export class LeaseBusyError extends Error {
|
||||
readonly userId: string
|
||||
readonly holder: string | null
|
||||
readonly leaseUntil: number
|
||||
constructor(userId: string, holder: string | null, leaseUntil: number) {
|
||||
super(`instance ${userId} is held by ${holder ?? 'someone'} until ${new Date(leaseUntil).toISOString()}`)
|
||||
this.name = 'LeaseBusyError'
|
||||
this.userId = userId
|
||||
this.holder = holder
|
||||
this.leaseUntil = leaseUntil
|
||||
}
|
||||
}
|
||||
|
||||
export interface LeasedSpawnerOptions extends LeaseOptions {
|
||||
/** 本机(= 它所属的 worker)在 `dsh_hosts.id` 里的标识。 */
|
||||
hostId: string
|
||||
/** worker agent 基址。 */
|
||||
agentUrl: string
|
||||
/** 与 agent 约定的共享密钥。 */
|
||||
agentToken: string
|
||||
/** 心跳间隔(ms)。默认 = 续租间隔。 */
|
||||
heartbeatMs?: number
|
||||
/** 该 worker 的内存预算(MB);**0 = 不承载实例**(只做门户/控制,设计 §15.3)。 */
|
||||
capacityMb?: number
|
||||
/**
|
||||
* **选机**(T08 S6):返回这次要把实例放到的 `hostId`。
|
||||
*
|
||||
* ⚠️ 实现必须**先粘性、再容量**(2026-09-15 生产切换时补的设计缺口):
|
||||
* 用户的工作区是**跟机器走的**(本地盘)⇒ 把"已有历史数据在某台"的用户调度到另一台,
|
||||
* 他打开实例会看到**空工作区**。所以:有历史归属且那台还 `up` ⇒ **留在原地**;
|
||||
* 只有"从没有过归属"(新用户)才按容量挑最空的。
|
||||
* 传入 `userId` 就是为了让实现能做这件事。
|
||||
*/
|
||||
selectHost?: (userId?: string) => Promise<string | undefined>
|
||||
/** **hostId → agent 地址/密钥**(多机时 fence 要发给"实例所在的那台")。 */
|
||||
agentFor?: (hostId: string) => { agentUrl: string; token: string } | undefined
|
||||
/**
|
||||
* 启动时把自己注册进 `dsh_hosts`(幂等)。
|
||||
*
|
||||
* ⚠️ **一个 agent 只应有一条 host 记录**:`registerSelf` 只在"本 Manager 与 worker 同机"
|
||||
* (1a 形态)时该开。**专用 Manager 部署必须关掉**(`DSHS_CLUSTER_REGISTER_SELF=0`),
|
||||
* 否则会多出一条指向同一 agent 的 host 记录 ⇒ 同一个用户可能被两个 hostId 各自认领。
|
||||
*/
|
||||
registerSelf?: boolean
|
||||
/** 关闭心跳(测试里手动 tick 用)。 */
|
||||
manual?: boolean
|
||||
}
|
||||
|
||||
/** 心跳里上报给 `dsh_hosts` 的本机观测值。 */
|
||||
export interface HostObservation {
|
||||
ok: boolean
|
||||
instances: number
|
||||
lastError?: string
|
||||
}
|
||||
|
||||
export class LeasedSpawner implements Spawner {
|
||||
private readonly lease: InstanceLease
|
||||
private timer: NodeJS.Timeout | undefined
|
||||
private lastObservation: HostObservation | undefined
|
||||
private readonly heartbeatMs: number
|
||||
|
||||
constructor(
|
||||
private readonly inner: Spawner,
|
||||
private readonly db: DbAdapter,
|
||||
private readonly options: LeasedSpawnerOptions,
|
||||
) {
|
||||
this.lease = new InstanceLease(db, options.hostId, options)
|
||||
this.heartbeatMs = options.heartbeatMs ?? this.lease.renewMs
|
||||
}
|
||||
|
||||
get hostId(): string {
|
||||
return this.options.hostId
|
||||
}
|
||||
|
||||
/** 最近一次心跳观测(管理面/诊断用)。 */
|
||||
observation(): HostObservation | undefined {
|
||||
return this.lastObservation
|
||||
}
|
||||
|
||||
/** 本 Manager 当前持有的实例(userId → {epoch, hostId})。 */
|
||||
holdings(): ReadonlyMap<string, { epoch: number; hostId: string }> {
|
||||
return this.lease.holdings()
|
||||
}
|
||||
|
||||
/** 注册本机 + 起心跳。窗口未开时先注册一次(否则管理面看不到这台 worker)。 */
|
||||
async start(): Promise<void> {
|
||||
if (this.options.registerSelf !== false) {
|
||||
await this.db.upsertDshHost({
|
||||
id: this.options.hostId,
|
||||
endpoint: this.options.agentUrl,
|
||||
agentToken: this.options.agentToken,
|
||||
capacityMb: this.options.capacityMb ?? 0,
|
||||
})
|
||||
}
|
||||
await this.tick() // 立即一次,管理面马上能看到心跳
|
||||
if (this.options.manual === true) return
|
||||
this.timer = setInterval(() => void this.tick(), this.heartbeatMs)
|
||||
this.timer.unref?.()
|
||||
}
|
||||
|
||||
/** 只停**心跳定时器**(不改实例)—— 注意别和 `Spawner.stop(userId)` 混淆,故另起名。 */
|
||||
stopHeartbeat(): void {
|
||||
if (this.timer !== undefined) {
|
||||
clearInterval(this.timer)
|
||||
this.timer = undefined
|
||||
}
|
||||
}
|
||||
|
||||
/** 一次心跳:续租 → 失权则 fence → 上报本机状态。 */
|
||||
async tick(): Promise<void> {
|
||||
// ① 续租;失权的 userId 会被清出本地记录
|
||||
const lost = await this.lease.renewAll()
|
||||
for (const userId of lost) {
|
||||
// ② 我已失权 ⇒ 让 worker 停掉那个实例(下发的 epoch 取 DB 当前值 +1,确保高于它的记录)
|
||||
await this.fenceOnAgent(userId)
|
||||
}
|
||||
await this.reportHost()
|
||||
}
|
||||
|
||||
private async fenceOnAgent(userId: string): Promise<void> {
|
||||
try {
|
||||
const inst = await this.db.findUserInstance(userId, 'main')
|
||||
// 多机(T08 S6):必须发给**实例所在的那台** —— 发错 host 等于没拦(旧持有者继续写)
|
||||
const target = inst?.hostId === null || inst?.hostId === undefined
|
||||
? { agentUrl: this.options.agentUrl, token: this.options.agentToken }
|
||||
: (this.options.agentFor?.(inst.hostId) ?? { agentUrl: this.options.agentUrl, token: this.options.agentToken })
|
||||
await this.post(`/fence`, { userId, epoch: (inst?.epoch ?? 0) + 1 }, target)
|
||||
} catch {
|
||||
// 通知失败不抛:下一轮心跳会重试;即便一直失败,租约已过期 ⇒ 新持有者会重建实例
|
||||
}
|
||||
}
|
||||
|
||||
private async reportHost(): Promise<void> {
|
||||
const base = this.options.agentUrl.replace(/\/$/, '')
|
||||
try {
|
||||
const res = await fetch(`${base}/healthz`, { signal: AbortSignal.timeout(this.heartbeatMs) })
|
||||
const body = (await res.json()) as { ok?: boolean; instances?: number }
|
||||
this.lastObservation = { ok: body.ok === true, instances: body.instances ?? 0 }
|
||||
await this.db.setDshHostStatus(
|
||||
this.options.hostId,
|
||||
this.lastObservation.ok ? 'up' : 'down',
|
||||
undefined,
|
||||
Date.now(),
|
||||
)
|
||||
} catch (err) {
|
||||
this.lastObservation = { ok: false, instances: 0, lastError: err instanceof Error ? err.message : String(err) }
|
||||
await this.db.setDshHostStatus(this.options.hostId, 'down', undefined, Date.now())
|
||||
}
|
||||
}
|
||||
|
||||
private async post(
|
||||
path: string,
|
||||
body: Record<string, unknown>,
|
||||
target?: { agentUrl: string; token: string },
|
||||
): Promise<unknown> {
|
||||
const use = target ?? { agentUrl: this.options.agentUrl, token: this.options.agentToken }
|
||||
const base = use.agentUrl.replace(/\/$/, '')
|
||||
const res = await fetch(`${base}${path}`, {
|
||||
method: 'POST',
|
||||
headers: { [AGENT_TOKEN_HEADER]: use.token, 'content-type': 'application/json' },
|
||||
body: JSON.stringify(body),
|
||||
signal: AbortSignal.timeout(10_000),
|
||||
})
|
||||
if (!res.ok) throw new Error(`agent POST ${path} → ${res.status}`)
|
||||
return res.json()
|
||||
}
|
||||
|
||||
// ── Spawner 实现 ────────────────────────────────────────────────────────
|
||||
|
||||
async launch(
|
||||
userId: string,
|
||||
folder: string,
|
||||
patch?: string,
|
||||
opts?: { force?: boolean; epoch?: number; hostId?: string },
|
||||
): Promise<Instance> {
|
||||
// **先选机、再认领**(T08 S6):租约的 host_id 必须与"实例真正落在哪台"一致,
|
||||
// 否则续租/释放会指向错误的对象(多机下就是静默的脑裂入口)。
|
||||
// 显式 `opts.hostId`(迁移的目标机)优先于自动选机。
|
||||
const hostId = opts?.hostId ?? (await this.options.selectHost?.(userId)) ?? this.options.hostId
|
||||
const claim = await this.lease.acquire(userId, hostId, { folder, patch })
|
||||
if (!claim.ok) throw new LeaseBusyError(userId, claim.holder, claim.leaseUntil)
|
||||
try {
|
||||
return await this.inner.launch(userId, folder, patch, { ...opts, epoch: claim.epoch, hostId })
|
||||
} catch (err) {
|
||||
// 拉起失败就**立刻交还归属** —— 否则要白等一个 TTL 才能重试(用户侧表现为"卡住")
|
||||
await this.lease.release(userId)
|
||||
throw err
|
||||
}
|
||||
}
|
||||
|
||||
async restartMain(userId: string): Promise<Instance | undefined> {
|
||||
const current = await this.inner.status(userId)
|
||||
if (current.main === undefined) return undefined
|
||||
const { folder, patch } = current.main
|
||||
await this.stop(userId)
|
||||
return this.launch(userId, folder, patch)
|
||||
}
|
||||
|
||||
/** 只重启**本 Manager 持有**的实例 —— 别人的归属不该被我重启(会与其持有者抢同一个 home)。 */
|
||||
async restartAllMains(): Promise<void> {
|
||||
for (const userId of [...this.lease.holdings().keys()]) {
|
||||
try {
|
||||
await this.restartMain(userId)
|
||||
} catch {
|
||||
// 单个失败不打断其余
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async spawnWatchdog(userId: string): Promise<Instance | undefined> {
|
||||
return this.inner.spawnWatchdog(userId)
|
||||
}
|
||||
|
||||
async status(userId: string): Promise<UserStatus> {
|
||||
return this.inner.status(userId)
|
||||
}
|
||||
|
||||
async endpointFor(userId: string): Promise<Endpoint | undefined> {
|
||||
return this.inner.endpointFor(userId)
|
||||
}
|
||||
|
||||
async stop(userId: string, hostId?: string): Promise<void> {
|
||||
// 显式 host 优先;否则用**我认领时那台**(认领记录里有)—— 别让 stop 落到别的 worker 上
|
||||
const target = hostId ?? this.lease.holdings().get(userId)?.hostId
|
||||
await this.inner.stop(userId, target)
|
||||
await this.lease.release(userId)
|
||||
}
|
||||
|
||||
async teardown(): Promise<void> {
|
||||
this.stopHeartbeat()
|
||||
await this.inner.teardown()
|
||||
}
|
||||
|
||||
async waitForLaunchTokenForUser(userId: string, timeoutMs?: number): Promise<void> {
|
||||
return this.inner.waitForLaunchTokenForUser(userId, timeoutMs)
|
||||
}
|
||||
|
||||
async restartAndProbe(userId: string, settleMs?: number): Promise<{ ok: boolean; reason: string }> {
|
||||
return this.inner.restartAndProbe(userId, settleMs)
|
||||
}
|
||||
|
||||
touch(userId: string): void {
|
||||
this.inner.touch(userId)
|
||||
}
|
||||
|
||||
async ensureFileService(userId: string): Promise<void> {
|
||||
return this.inner.ensureFileService(userId)
|
||||
}
|
||||
|
||||
/**
|
||||
* 透传可选观测面。`Spawner` 里这两个是**可选**方法 ⇒ 这里做条件委托:
|
||||
* 内层有就转发(熔断/配额是 worker 本地自管的概念,设计 §1.2),没有就回 null。
|
||||
*/
|
||||
breakerInfo(userId: string): { opens: number; openedAt: number; cooldownUntil: number } | null {
|
||||
return this.inner.breakerInfo?.(userId) ?? null
|
||||
}
|
||||
|
||||
quotaInfo(userId: string): { baseMb: number; memMb: number; heapMb: number } | null {
|
||||
return this.inner.quotaInfo?.(userId) ?? null
|
||||
}
|
||||
}
|
||||
@@ -22,7 +22,7 @@ import {
|
||||
realpathSync,
|
||||
writeFileSync,
|
||||
} from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { dirname, join } from 'node:path'
|
||||
import type { ServerConfig } from '../config.js'
|
||||
import { handoffPath, homeRoot, userRoot, workspaceRoot } from '../fs/workspace.js'
|
||||
import {
|
||||
@@ -129,6 +129,47 @@ function withHeap(base: string | undefined, memMb: number): string {
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* 列出 `dest` 与 `stopAt` 之间的**祖先目录**(由外到内),用于 bwrap 的 `--tmpfs`(见
|
||||
* {@link mountParentDirArgs}:`--tmpfs` 自带 0755,且**兼容 47 上的 bwrap 0.4.0**)。
|
||||
*
|
||||
* 背景(T08 S1.6,2026-09-15 实测):bwrap **只创建挂载点本身**,沿途缺失的父目录由它自建,
|
||||
* 而权限是 **`0700 root:root`** —— 实测 `--bind /opt/a/b/c /opt/a/b/c` 会得到 `/opt`、
|
||||
* `/opt/a`、`/opt/a/b` **全是 0700**。后果:**实例以非 root 的 uid 穿越这些路径时 EACCES**。
|
||||
* 已实测到的两处症状:
|
||||
* ① 宿主上 `/etc/ssl/openssl.cnf` 是指向 `/etc/pki/tls/openssl.cnf` 的**符号链接** ⇒ 解析要穿过
|
||||
* `/etc/pki`(0700)⇒ node 报 `OpenSSL configuration error … Permission denied`、**exitCode 13**
|
||||
* (106 / OpenCloudOS 9.6 实测;47 上**没有**该文件故静默跳过 ⇒ 同一份代码一台能跑一台崩);
|
||||
* ② **用户工作区在沙箱内不可穿越** ⇒ 实例按**绝对路径**读写自己的文件被拒。
|
||||
* 修法:把这些祖先目录**显式建成 0755**。**权限不扩大** —— 这些目录里只有随后绑定的白名单内容
|
||||
* (整绑 `/etc/pki` 的替代方案已否决:会带入 `/etc/pki/tls/private/postfix.key`,违反 R5)。
|
||||
*/
|
||||
function mountParentDirList(dest: string, stopAt: string): string[] {
|
||||
const dirs: string[] = []
|
||||
let cur = dirname(dest)
|
||||
while (cur !== stopAt && cur !== '/' && cur !== '' && cur !== '.') {
|
||||
dirs.push(cur)
|
||||
cur = dirname(cur)
|
||||
}
|
||||
return dirs.reverse() // 由外到内(`--perms` 只作用于紧接着的那一个 `--dir`)
|
||||
}
|
||||
|
||||
/**
|
||||
* 把 {@link mountParentDirList} 的结果摊平成 bwrap 参数。
|
||||
* ⚠️ **不要再"就近调用"它**(例如插在 `--bind` 之前)—— 见 `bwrapArgs` 里"统一前置"
|
||||
* 那段注释:就近创建会在嵌套前缀下遮掉已绑好的挂载点。当前实现只在**一处**统一使用。
|
||||
*/
|
||||
function mountParentDirArgs(dest: string, stopAt: string): string[] {
|
||||
// ⚠️ **必须用 `--tmpfs`,不能用 `--perms 0755 --dir`**(2026-09-15 实测,差点打断生产):
|
||||
// · `--perms` 是 bubblewrap **0.5+** 才有的选项;**47 上是 0.4.0** ⇒ 传了直接
|
||||
// `bwrap: Unknown option --perms` ⇒ **沙箱起不来 = 所有实例全挂**(106 是 0.11.0,能过)。
|
||||
// · 而 `--tmpfs` **自带 0755**(本文件另一处注释也这么写:`bwrap 的 --tmpfs 权限是 755`),
|
||||
// 且在 0.4.0 上就可用。
|
||||
// 47 上实测:改用 `--tmpfs` 后 `/etc` 可见条目 **78 → 78(零变化)**,且 `/etc/pki` 权限
|
||||
// 由 `drwx------` 变为 `drwxr-xr-x`(可穿越)。代价 = 每个中间目录多一个空 tmpfs 挂载(极小)。
|
||||
return mountParentDirList(dest, stopAt).flatMap((d) => ['--tmpfs', d])
|
||||
}
|
||||
|
||||
/**
|
||||
* Local backend: owns the lifecycle of per-user DSH process pairs via
|
||||
* child_process. State is in-memory. Implements {@link Spawner}.
|
||||
@@ -361,6 +402,26 @@ export class LocalSpawner implements Spawner {
|
||||
return { main: this.mains.get(userId), watchdog: this.watchdogs.get(userId) }
|
||||
}
|
||||
|
||||
/**
|
||||
* 整机视角的实例清单(T08 S3:worker agent 的 `/instances` 用)。
|
||||
*
|
||||
* 口径 = **每个用户的 main 实例**(watchdog 是一次性 headless,不进对账口径)。
|
||||
* 设计上这是 `/instances` "一次拿回整机"的实现,替代逐用户查询(设计 §11.6)。
|
||||
*/
|
||||
listUserInstances(): Instance[] {
|
||||
return [...this.mains.values()]
|
||||
}
|
||||
|
||||
/**
|
||||
* 当前 main 实例的 launch token(T08 S3 P0-6)。
|
||||
*
|
||||
* 本地模式下 token 从子进程 stdout 解析出来;跨机后 **agent 必须把它回传 Manager**,
|
||||
* 否则「登录直达会话」(档案 06/13/15)与实例侧 401 自愈(档案 24/50/51)都会失效。
|
||||
*/
|
||||
launchTokenOf(userId: string): string | undefined {
|
||||
return this.mains.get(userId)?.launchToken
|
||||
}
|
||||
|
||||
/** Endpoint the proxy forwards to (local → the running main's loopback port). */
|
||||
async endpointFor(userId: string): Promise<Endpoint | undefined> {
|
||||
const port = this.mains.get(userId)?.port
|
||||
@@ -680,6 +741,30 @@ export class LocalSpawner implements Spawner {
|
||||
'/etc/ssl',
|
||||
]
|
||||
const out: string[] = []
|
||||
// 2026-09-15(T08 S1.6)**中间挂载点必须可穿越(0755)**。
|
||||
//
|
||||
// 现象(106 / OpenCloudOS 9.6 实测):实例起不来,子进程报
|
||||
// `/usr/bin/node: OpenSSL configuration error: … Permission denied:
|
||||
// … fopen(/etc/ssl/openssl.cnf, rb)` ⇒ **exitCode 13**。
|
||||
// 根因:bwrap 会为 `--ro-bind-try /etc/pki/tls/certs …` 这类路径**自动补齐父目录**,
|
||||
// 而这些自动创建的目录权限是 **0700(drwx------ root:root)** ⇒ 非 root 的实例
|
||||
// **无法穿越**;宿主上 `/etc/ssl/openssl.cnf` 恰好是**指向 `/etc/pki/tls/openssl.cnf`
|
||||
// 的符号链接** ⇒ 解析要穿过 `/etc/pki` → 被拒 → 报 **EACCES(不是 ENOENT)**
|
||||
// → node 读 OpenSSL 配置**硬失败**。
|
||||
// 为什么 47 没事:Alibaba Cloud Linux 3 上**没有** `/etc/ssl/openssl.cnf`
|
||||
// ⇒ node 静默跳过 ⇒ **同一份代码一台能跑、一台崩**(机器基线差异,设计 §14.3)。
|
||||
// 修法:在绑定**之前**把白名单路径在 `/etc` 下的所有中间目录显式建成 0755。
|
||||
// 权限**不扩大**:`/etc` 在本沙箱里是 tmpfs,这些目录里只有下面白名单绑定的内容,
|
||||
// 不新增任何宿主可见面。("整绑 `/etc/pki`"的替代方案已否决 —— 会顺带带入
|
||||
// `/etc/pki/tls/private/postfix.key`,违反 **R5 权限只准收窄**。)
|
||||
// 注意顺序:由外到内(内层挂载点要求外层已存在)。
|
||||
const intermediates = new Set<string>()
|
||||
for (const p of allow) {
|
||||
for (const d of mountParentDirList(p, '/etc')) intermediates.add(d)
|
||||
}
|
||||
for (const d of [...intermediates].sort((a, b) => a.split('/').length - b.split('/').length)) {
|
||||
out.push('--tmpfs', d) // 见 mountParentDirArgs 的注释:`--tmpfs` 自带 0755 且兼容 bwrap 0.4.0
|
||||
}
|
||||
for (const p of allow) {
|
||||
let src = p
|
||||
try {
|
||||
@@ -691,6 +776,23 @@ export class LocalSpawner implements Spawner {
|
||||
}
|
||||
return out
|
||||
})(),
|
||||
// ── 所有挂载点的**中间目录**统一在这里建好(T08 S1.6 修正版)────────────────
|
||||
// 为什么必须"统一前置 + 去重 + 由外到内"(2026-09-15 实测踩到的真 bug):
|
||||
// 用户根(`--bind root root`)与共享技能层(`--ro-bind-try skill skill`)**可能嵌套在
|
||||
// 同一前缀下**。若按"就近创建"把技能层的中间目录插在 `--bind root root` **之后**,
|
||||
// 那么后挂的 `--tmpfs <共同祖先>` 会把**已经绑好的用户根整个遮掉** ⇒ bwrap 报
|
||||
// `Can't chdir to <userRoot>/ws/xxx: No such file or directory` ⇒ 实例崩溃循环。
|
||||
// 前置 + 去重后,中间目录只建一次,后续所有 bind 都落在它里面,谁也不遮谁。
|
||||
// 权限不扩大:这些目录里只有随后绑定的白名单内容。
|
||||
...(() => {
|
||||
const dirs = new Set<string>()
|
||||
for (const dest of [root, this.config.bundledSkillDir].filter((d) => d !== '')) {
|
||||
for (const d of mountParentDirList(dest, '/')) dirs.add(d)
|
||||
}
|
||||
return [...dirs]
|
||||
.sort((a, b) => a.split('/').length - b.split('/').length)
|
||||
.flatMap((d) => ['--tmpfs', d])
|
||||
})(),
|
||||
'--dev', '/dev', '--proc', '/proc',
|
||||
'--bind', tmpDir, '/tmp',
|
||||
'--bind', root, root,
|
||||
|
||||
@@ -0,0 +1,330 @@
|
||||
/**
|
||||
* `Spawner` 的**远端实现**(T08 S3 单机 / S6 多机;设计 §1.1/§11)。
|
||||
*
|
||||
* 路由层只依赖 `Spawner` 接口(见 `spawner.ts` 头注释),所以本类**不触碰路由与代理**
|
||||
* —— 代理层把 `endpointFor` 返回的 `{host, port}` 直连即可(`proxy.ts` 的 TCP 目标
|
||||
* 与 Host 头本就是分开处理的,跨机不需要改信任逻辑)。
|
||||
*
|
||||
* 三条协议纪律:
|
||||
* 1. **幂等键复用**:同一次调用的重试**复用同一个 `operationId`** —— 否则 agent 会把
|
||||
* "Manager 超时后重发"当成新请求,起出两个实例(设计 §11.3 手段 3)。
|
||||
* 2. **重试有界**:3 次(200ms / 1s / 3s),仍失败就**抛错**,由上层决定退让或告警;
|
||||
* 错误信息里带上状态码与响应体片段,避免"静默失败"。
|
||||
* 3. **按 host 路由**(S6):每个用户的操作都落到**它实例所在的那台** —— 依据是
|
||||
* `dsh_instances.host_id`,由上层以 `hostIdFor` 注入(本类不直接连 DB)。
|
||||
*
|
||||
* ⚠️ 本类**不持有归属租约**:租约由 Manager 侧的 `InstanceLease`/`LeasedSpawner` 管理。
|
||||
*
|
||||
* @module dshs/supervisor/remote-spawner
|
||||
*/
|
||||
import { randomUUID } from 'node:crypto'
|
||||
import { AGENT_TOKEN_HEADER } from '../worker/agent.js'
|
||||
import type { Endpoint, Instance, Spawner, UserStatus } from './spawner.js'
|
||||
|
||||
/** 一台 worker 的接入信息。 */
|
||||
export interface ClusterHost {
|
||||
hostId: string
|
||||
/** agent 基址。 */
|
||||
agentUrl: string
|
||||
/** 与 agent 约定的共享密钥。 */
|
||||
token: string
|
||||
/** 代理时使用的主机(同机 1a = `127.0.0.1`;跨机填 Worker 内网 IP)。 */
|
||||
instanceHost?: string
|
||||
}
|
||||
|
||||
export interface RemoteSpawnerOptions {
|
||||
/** agent 基址,如 `http://127.0.0.1:9000`。 */
|
||||
agentUrl: string
|
||||
/** 与 agent 约定的共享密钥。 */
|
||||
token: string
|
||||
/** 代理时使用的主机(同机 1a = `127.0.0.1`;跨机填 Worker 内网 IP)。 */
|
||||
instanceHost?: string
|
||||
/** 单次请求超时(ms)。控制通道是短请求,默认 10 s(设计 §11.2)。 */
|
||||
timeoutMs?: number
|
||||
/** 注入 fetch(测试用)。 */
|
||||
fetchImpl?: typeof fetch
|
||||
/**
|
||||
* 解析该用户的模型 key 与 uid —— Manager 侧解析后**随 launch 投递**给 agent。
|
||||
* 为什么不投递"让 Worker 自己查凭据库":那是**权限扩大**(Worker 就能读全量用户的 key),
|
||||
* 而投递是收窄到"本次实例那一把"。**注意**:这条只管**控制面凭据**,与"Worker 能不能有
|
||||
* 自己的库"无关(插件数据在实例 home 里,见设计 §1.3 数据分层)。
|
||||
*/
|
||||
resolveApiKey?: (userId: string) => Promise<string | null>
|
||||
resolveUid?: (userId: string) => Promise<number>
|
||||
/** 默认 host 的 id(不提供 `hosts` 时的单机形态用它)。 */
|
||||
defaultHostId?: string
|
||||
/** **多机(S6)**:除默认 host 外的其它 worker。给了就按 `hostId` 路由。 */
|
||||
hosts?: ClusterHost[]
|
||||
/** **多机(S6)**:`userId` → 它实例所在的 `hostId`(上层查 `dsh_instances.host_id` 注入)。 */
|
||||
hostIdFor?: (userId: string) => Promise<string | undefined>
|
||||
/**
|
||||
* **host 目录的来源**(S6):从 `dsh_hosts` 读。给它就**不必预知 worker 列表**,
|
||||
* 且**新增 worker 无需重启 Manager**(TTL 内自动生效)。
|
||||
*/
|
||||
hostsProvider?: () => Promise<ClusterHost[]>
|
||||
/** 目录缓存时长(ms)。默认 30 s —— 与心跳同量级。 */
|
||||
directoryTtlMs?: number
|
||||
}
|
||||
|
||||
const RETRY_DELAYS_MS = [200, 1000, 3000]
|
||||
|
||||
export class RemoteSpawner implements Spawner {
|
||||
private readonly defaultHost: ClusterHost
|
||||
private readonly hosts = new Map<string, ClusterHost>()
|
||||
private readonly timeoutMs: number
|
||||
private readonly doFetch: typeof fetch
|
||||
private readonly resolveApiKey?: (userId: string) => Promise<string | null>
|
||||
private readonly resolveUid?: (userId: string) => Promise<number>
|
||||
private readonly hostIdFor?: (userId: string) => Promise<string | undefined>
|
||||
private readonly hostsProvider?: () => Promise<ClusterHost[]>
|
||||
private readonly directoryTtlMs: number
|
||||
private directoryLoadedAt = 0
|
||||
|
||||
constructor(options: RemoteSpawnerOptions) {
|
||||
this.defaultHost = {
|
||||
hostId: options.defaultHostId ?? 'local',
|
||||
agentUrl: options.agentUrl.replace(/\/$/, ''),
|
||||
token: options.token,
|
||||
instanceHost: options.instanceHost ?? '127.0.0.1',
|
||||
}
|
||||
this.hosts.set(this.defaultHost.hostId, this.defaultHost)
|
||||
for (const host of options.hosts ?? []) {
|
||||
this.hosts.set(host.hostId, { ...host, agentUrl: host.agentUrl.replace(/\/$/, '') })
|
||||
}
|
||||
this.timeoutMs = options.timeoutMs ?? 10_000
|
||||
this.doFetch = options.fetchImpl ?? fetch
|
||||
this.resolveApiKey = options.resolveApiKey
|
||||
this.resolveUid = options.resolveUid
|
||||
this.hostIdFor = options.hostIdFor
|
||||
this.hostsProvider = options.hostsProvider
|
||||
this.directoryTtlMs = options.directoryTtlMs ?? 30_000
|
||||
}
|
||||
|
||||
/**
|
||||
* 按需刷新 host 目录(TTL 内不重复查询)。
|
||||
* **默认 host 始终在表里**(配置里那台),即使它还没注册进 `dsh_hosts`。
|
||||
*/
|
||||
private async ensureHosts(): Promise<void> {
|
||||
if (this.hostsProvider === undefined) return
|
||||
if (Date.now() - this.directoryLoadedAt < this.directoryTtlMs) return
|
||||
this.directoryLoadedAt = Date.now()
|
||||
try {
|
||||
for (const host of await this.hostsProvider()) {
|
||||
this.hosts.set(host.hostId, { ...host, agentUrl: host.agentUrl.replace(/\/$/, '') })
|
||||
}
|
||||
this.hosts.set(this.defaultHost.hostId, this.defaultHost)
|
||||
} catch {
|
||||
// 查库失败就沿用旧目录(可能是全库不可用的前兆,由心跳/管理面暴露)
|
||||
}
|
||||
}
|
||||
|
||||
/** 已知的 host 目录(管理面/诊断用)。 */
|
||||
knownHosts(): ClusterHost[] {
|
||||
return [...this.hosts.values()]
|
||||
}
|
||||
|
||||
/** 强制刷新目录(管理面/测试用)。 */
|
||||
async reloadHosts(): Promise<void> {
|
||||
this.directoryLoadedAt = 0
|
||||
await this.ensureHosts()
|
||||
}
|
||||
|
||||
/** hostId → 接入信息;未知 host 回退到默认(单机形态下这就是唯一那台)。 */
|
||||
hostById(hostId: string | null | undefined): ClusterHost {
|
||||
if (hostId === null || hostId === undefined) return this.defaultHost
|
||||
return this.hosts.get(hostId) ?? this.defaultHost
|
||||
}
|
||||
|
||||
/**
|
||||
* 决定这次操作落到哪台。优先级:**显式指定**(迁移目标机、launch 时选好的机)
|
||||
* → **该用户实例的归属**(`hostIdFor`)→ 默认 host。
|
||||
*/
|
||||
private async hostFor(userId: string, explicit?: string): Promise<ClusterHost> {
|
||||
await this.ensureHosts()
|
||||
if (explicit !== undefined) {
|
||||
let found = this.hosts.get(explicit)
|
||||
if (found === undefined) {
|
||||
// 目录有 30s TTL:显式指定的 host 可能"刚 join 还没进目录" ⇒ 强制刷一次
|
||||
this.directoryLoadedAt = 0
|
||||
await this.ensureHosts()
|
||||
found = this.hosts.get(explicit)
|
||||
}
|
||||
// ⛔ 仍然找不到就**报错**,绝不回退到默认 host ——
|
||||
// "租约认领在 A、实例却起在 B"是多机下最危险的静默失败(归属与实例分离)。
|
||||
if (found === undefined) throw new Error(`unknown host "${explicit}":不在 host 目录里,拒绝改投到别的 worker`)
|
||||
return found
|
||||
}
|
||||
if (this.hostIdFor !== undefined) {
|
||||
const owned = await this.hostIdFor(userId)
|
||||
if (owned !== undefined) {
|
||||
const found = this.hosts.get(owned)
|
||||
if (found !== undefined) return found
|
||||
}
|
||||
}
|
||||
return this.defaultHost
|
||||
}
|
||||
|
||||
/** 带重试的请求。`operationId` 由调用方生成并在重试间**保持不变**(幂等)。 */
|
||||
private async call<T>(
|
||||
host: ClusterHost,
|
||||
method: 'GET' | 'POST',
|
||||
path: string,
|
||||
body?: Record<string, unknown>,
|
||||
operationId?: string,
|
||||
): Promise<T> {
|
||||
const payload = body === undefined ? undefined : { ...body, ...(operationId === undefined ? {} : { operationId }) }
|
||||
let lastErr: unknown
|
||||
for (let attempt = 0; attempt <= RETRY_DELAYS_MS.length; attempt += 1) {
|
||||
if (attempt > 0) await new Promise((r) => setTimeout(r, RETRY_DELAYS_MS[attempt - 1]))
|
||||
try {
|
||||
const res = await this.doFetch(`${host.agentUrl}${path}`, {
|
||||
method,
|
||||
headers: {
|
||||
[AGENT_TOKEN_HEADER]: host.token,
|
||||
...(payload === undefined ? {} : { 'content-type': 'application/json' }),
|
||||
},
|
||||
body: payload === undefined ? undefined : JSON.stringify(payload),
|
||||
signal: AbortSignal.timeout(this.timeoutMs),
|
||||
})
|
||||
if (res.ok) return (await res.json()) as T
|
||||
const text = await res.text()
|
||||
// 4xx 是"协议/参数错",重试没意义;5xx 与网络错才重试。
|
||||
if (res.status < 500) throw new Error(`agent ${method} ${path} → ${res.status}: ${text.slice(0, 200)}`)
|
||||
lastErr = new Error(`agent ${method} ${path} → ${res.status}: ${text.slice(0, 200)}`)
|
||||
} catch (err) {
|
||||
lastErr = err
|
||||
}
|
||||
}
|
||||
throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))
|
||||
}
|
||||
|
||||
async launch(
|
||||
userId: string,
|
||||
folder: string,
|
||||
patch?: string,
|
||||
opts?: { force?: boolean; epoch?: number; hostId?: string },
|
||||
): Promise<Instance> {
|
||||
const host = await this.hostFor(userId, opts?.hostId)
|
||||
// 同一个 operationId 贯穿这次调用的所有重试 ⇒ agent 侧幂等回放(不会起两个实例)。
|
||||
const operationId = randomUUID()
|
||||
const apiKey = this.resolveApiKey === undefined ? null : await this.resolveApiKey(userId)
|
||||
const uid = this.resolveUid === undefined ? undefined : await this.resolveUid(userId)
|
||||
const res = await this.call<{ instance: Instance; note?: string }>(
|
||||
host,
|
||||
'POST',
|
||||
'/launch',
|
||||
{ userId, folder, patch, apiKey, uid, epoch: opts?.epoch },
|
||||
operationId,
|
||||
)
|
||||
return res.instance
|
||||
}
|
||||
|
||||
async restartMain(userId: string, hostId?: string): Promise<Instance | undefined> {
|
||||
const current = await this.status(userId)
|
||||
if (current.main === undefined) return undefined
|
||||
const { folder, patch } = current.main
|
||||
const host = await this.hostFor(userId, hostId)
|
||||
await this.stop(userId, host.hostId)
|
||||
return this.launch(userId, folder, patch, { hostId: host.hostId })
|
||||
}
|
||||
|
||||
/** 重启**所有 host 上**能看到的实例(跨机聚合;上层 `LeasedSpawner` 会限定在自己持有的范围内)。 */
|
||||
async restartAllMains(): Promise<void> {
|
||||
for (const host of this.hosts.values()) {
|
||||
let instances: Array<{ userId: string }> = []
|
||||
try {
|
||||
const res = await this.call<{ instances: Array<{ userId: string }> }>(host, 'GET', '/instances')
|
||||
instances = res.instances
|
||||
} catch {
|
||||
continue // 该 host 不可达:跳过(心跳/告警负责暴露)
|
||||
}
|
||||
for (const inst of instances) {
|
||||
try {
|
||||
await this.restartMain(inst.userId, host.hostId)
|
||||
} catch {
|
||||
// 单台失败不打断其余(与 LocalSpawner 的语义一致)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async spawnWatchdog(userId: string): Promise<Instance | undefined> {
|
||||
const host = await this.hostFor(userId)
|
||||
const res = await this.call<{ instance: Instance | null }>(host, 'POST', `/watchdog/${encodeURIComponent(userId)}`)
|
||||
return res.instance ?? undefined
|
||||
}
|
||||
|
||||
async status(userId: string): Promise<UserStatus> {
|
||||
const host = await this.hostFor(userId)
|
||||
const res = await this.call<{ main: Instance | null }>(host, 'GET', `/status/${encodeURIComponent(userId)}`)
|
||||
return res.main === null ? {} : { main: res.main }
|
||||
}
|
||||
|
||||
/** 代理目标:由**实例所在那台** agent 给端口(未运行 → undefined,代理会走冷启动分支)。 */
|
||||
async endpointFor(userId: string): Promise<Endpoint | undefined> {
|
||||
const host = await this.hostFor(userId)
|
||||
const res = await this.call<{ running: boolean; host?: string; port?: number }>(
|
||||
host,
|
||||
'GET',
|
||||
`/endpoint/${encodeURIComponent(userId)}`,
|
||||
)
|
||||
return res.running && res.host !== undefined && res.port !== undefined
|
||||
? { host: res.host, port: res.port }
|
||||
: undefined
|
||||
}
|
||||
|
||||
async stop(userId: string, hostId?: string): Promise<void> {
|
||||
const host = await this.hostFor(userId, hostId)
|
||||
await this.call(host, 'POST', '/stop', { userId }, randomUUID())
|
||||
}
|
||||
|
||||
async teardown(): Promise<void> {
|
||||
// 远端实例的寿命长于任何单个 Manager 副本 ⇒ 由 Manager 的归属/租约管理,不在关闭时清。
|
||||
}
|
||||
|
||||
/**
|
||||
* 等 launch token 出现(本地模式 = 启动完成的信号)。
|
||||
* 跨机后 token 由 agent 回传,语义不变;超时返回(不抛)—— 与本地实现一致。
|
||||
*/
|
||||
async waitForLaunchTokenForUser(userId: string, timeoutMs = 20_000): Promise<void> {
|
||||
const deadline = Date.now() + timeoutMs
|
||||
while (Date.now() < deadline) {
|
||||
try {
|
||||
const st = await this.status(userId)
|
||||
if (st.main !== undefined && st.main.launchToken !== undefined) return
|
||||
if (st.main === undefined) return // 没实例/已停 ⇒ 立即返回(与本地实现一致)
|
||||
} catch {
|
||||
// 网络抖动:继续等
|
||||
}
|
||||
await new Promise((r) => setTimeout(r, 200))
|
||||
}
|
||||
}
|
||||
|
||||
async restartAndProbe(userId: string, settleMs?: number): Promise<{ ok: boolean; reason: string }> {
|
||||
const host = await this.hostFor(userId)
|
||||
return this.call<{ ok: boolean; reason: string }>(
|
||||
host,
|
||||
'POST',
|
||||
`/restart-probe/${encodeURIComponent(userId)}`,
|
||||
settleMs === undefined ? {} : { settleMs },
|
||||
)
|
||||
}
|
||||
|
||||
/** 活动信号:转发给**实例所在那台** agent,让它自己的 idle-reap 不误杀(fire-and-forget)。 */
|
||||
touch(userId: string): void {
|
||||
void this.hostFor(userId)
|
||||
.then((host) =>
|
||||
this.doFetch(`${host.agentUrl}/touch/${encodeURIComponent(userId)}`, {
|
||||
method: 'POST',
|
||||
headers: { [AGENT_TOKEN_HEADER]: host.token },
|
||||
}),
|
||||
)
|
||||
.catch(() => {
|
||||
/* 活动信号丢了不影响正确性 */
|
||||
})
|
||||
}
|
||||
|
||||
async ensureFileService(_userId: string): Promise<void> {
|
||||
// worker 本机就有用户卷(local 语义)⇒ 无需 sidecar。跨机文件面由 RemoteUserFs(/fs/*)承担。
|
||||
}
|
||||
}
|
||||
@@ -92,7 +92,22 @@ export interface Endpoint {
|
||||
* `LocalSpawner` materializes it to a file inside the user's own volume.
|
||||
*/
|
||||
export interface Spawner {
|
||||
launch(userId: string, folder: string, patch?: string, opts?: { force?: boolean }): Promise<Instance>
|
||||
/**
|
||||
* 拉起实例。
|
||||
*
|
||||
* `opts.epoch`(T08 S4):**集群模式下 epoch 是 launch 契约的一部分** —— Manager 先抢占
|
||||
* 归属拿到 epoch,再把它随 launch 下发,worker 记下来用于 **self-fencing**
|
||||
* (收到更高 epoch 就停掉自己那个实例)。本地模式忽略该字段。
|
||||
*
|
||||
* `opts.hostId`(T08 S6):**多 worker 时指定落到哪台** —— 由上层选好机、并已用它认领租约,
|
||||
* 因此这里必须与租约的 `host_id` 一致(否则归属与实例分离)。单机/1a 忽略。
|
||||
*/
|
||||
launch(
|
||||
userId: string,
|
||||
folder: string,
|
||||
patch?: string,
|
||||
opts?: { force?: boolean; epoch?: number; hostId?: string },
|
||||
): Promise<Instance>
|
||||
restartMain(userId: string): Promise<Instance | undefined>
|
||||
/**
|
||||
* 档案 78:熔断观测面(可选 —— 熔断是**本地模式**概念,k8s 模式没有)。
|
||||
@@ -112,7 +127,7 @@ export interface Spawner {
|
||||
spawnWatchdog(userId: string): Promise<Instance | undefined>
|
||||
status(userId: string): Promise<UserStatus>
|
||||
endpointFor(userId: string): Promise<Endpoint | undefined>
|
||||
stop(userId: string): Promise<void>
|
||||
stop(userId: string, hostId?: string): Promise<void>
|
||||
teardown(): Promise<void>
|
||||
/** 等待该用户 main 实例打印 launch token(本地模式 = 启动完成的信号)。无实例 /
|
||||
* 已崩溃 / 已停 → 立即返回;k8s 模式无 token 概念 → no-op。供 enter 复用分支在返回
|
||||
|
||||
@@ -155,4 +155,101 @@ export const adminRoutes: FastifyPluginAsync = async (app) => {
|
||||
}
|
||||
})
|
||||
|
||||
// ── 集群管理面(T08 S6):worker 注册表 + 实例迁移 ────────────────────────
|
||||
|
||||
/**
|
||||
* Worker 列表。
|
||||
* ⚠️ **绝不下发 `agentToken`** —— 它是内网共享密钥,只回 `hasToken` 供排查"配没配"。
|
||||
*/
|
||||
app.get('/api/admin/hosts', { preHandler: requireAdmin }, async () => {
|
||||
const hosts = await app.db.listDshHosts()
|
||||
return {
|
||||
deployMode: app.config.deployMode,
|
||||
hosts: hosts.map((h) => ({
|
||||
id: h.id,
|
||||
endpoint: h.endpoint,
|
||||
capacityMb: h.capacityMb,
|
||||
usedMb: h.usedMb,
|
||||
status: h.status,
|
||||
lastHeartbeat: h.lastHeartbeat,
|
||||
hasToken: h.agentToken !== '',
|
||||
})),
|
||||
}
|
||||
})
|
||||
|
||||
/** 注册/更新一台 worker(join 脚本调用;**幂等**:同 id 重复执行 = 更新并标回 `up`)。 */
|
||||
app.post('/api/admin/hosts', { preHandler: requireAdmin }, async (request, reply) => {
|
||||
const body = request.body as {
|
||||
id?: string
|
||||
endpoint?: string
|
||||
token?: string
|
||||
capacityMb?: number
|
||||
}
|
||||
if (body.id === undefined || body.endpoint === undefined || body.token === undefined) {
|
||||
return reply.code(400).send({ error: 'id, endpoint and token are required' })
|
||||
}
|
||||
const host = await app.db.upsertDshHost({
|
||||
id: body.id,
|
||||
endpoint: body.endpoint,
|
||||
agentToken: body.token,
|
||||
capacityMb: Number(body.capacityMb ?? 0),
|
||||
})
|
||||
await app.db.audit(request.user?.id ?? null, 'host.upsert', JSON.stringify({ id: host.id, endpoint: host.endpoint }))
|
||||
return {
|
||||
ok: true,
|
||||
host: { id: host.id, endpoint: host.endpoint, capacityMb: host.capacityMb, status: host.status },
|
||||
}
|
||||
})
|
||||
|
||||
/**
|
||||
* **计划内迁移**(T08 S6;设计 §4.2):drain → 目标机拉起 → 归属原子更新(epoch+1)。
|
||||
*
|
||||
* 顺序不可换:**先停源、再在目标机拉起**。如果反序,两台上会同时有实例(同一个 home ⇒ 双写)。
|
||||
* 归属的原子性由租约保证(`claimInstance` 会把 `host_id` 换成目标机并 `epoch+1`),
|
||||
* 所以旧机即便复活也会被 fencing 挡住(设计 §11.5)。
|
||||
*
|
||||
* ⚠️ **数据不搬家**:`folder` 是实例眼里的绝对路径,能在目标机上生效的前提是
|
||||
* **两台 worker 的 dataRoot 同路径 + 用户数据位置无关**(共享存储或已同步)——
|
||||
* 这正是设计 §12/§14.3 的前提,不是本路由能替你保证的。
|
||||
*/
|
||||
app.post('/api/admin/users/:id/dsh/migrate', { preHandler: requireAdmin }, async (request, reply) => {
|
||||
const { id } = request.params as { id: string }
|
||||
const { targetHost } = request.body as { targetHost?: string }
|
||||
if (targetHost === undefined || targetHost === '') {
|
||||
return reply.code(400).send({ error: 'targetHost is required' })
|
||||
}
|
||||
const before = await app.db.findUserInstance(id, 'main')
|
||||
if (before === undefined) return reply.code(404).send({ error: 'not_found', detail: '该用户没有 main 实例记录' })
|
||||
if (before.hostId === targetHost) return reply.code(409).send({ error: 'already_there' })
|
||||
|
||||
const target = await app.db.findDshHost(targetHost)
|
||||
if (target === undefined) return reply.code(404).send({ error: 'unknown_host' })
|
||||
if (target.status === 'down') return reply.code(409).send({ error: 'target_down' })
|
||||
|
||||
const source = before.hostId
|
||||
// fail-loud:没有 folder 就没法在目标机上复现启动(空 cwd 会让 bwrap 直接崩)
|
||||
if ((before.folder ?? '') === '') {
|
||||
return reply.code(409).send({ error: 'no_folder_recorded', detail: '该实例没有记录 folder,无法复现启动' })
|
||||
}
|
||||
// ① drain:停源机实例(优雅停机 → 会话落盘;同时释放归属)
|
||||
if (source !== null) await app.supervisor.stop(id, source)
|
||||
// ② 目标机拉起(走租约:以 targetHost 认领 → epoch+1)
|
||||
const instance = await app.supervisor.launch(id, before.folder ?? '', before.patch ?? undefined, {
|
||||
hostId: targetHost,
|
||||
})
|
||||
const after = await app.db.findUserInstance(id, 'main')
|
||||
await app.db.audit(
|
||||
request.user?.id ?? null,
|
||||
'dsh.migrate',
|
||||
JSON.stringify({ userId: id, from: source, to: after?.hostId, epoch: after?.epoch }),
|
||||
)
|
||||
return {
|
||||
ok: true,
|
||||
from: source,
|
||||
to: after?.hostId ?? null,
|
||||
epoch: after?.epoch ?? 0,
|
||||
port: instance.port ?? null,
|
||||
}
|
||||
})
|
||||
|
||||
}
|
||||
+144
-2
@@ -13,10 +13,13 @@ import { chown, mkdir, readFile, stat, writeFile } from 'node:fs/promises'
|
||||
import type { ServerConfig } from '../config.js'
|
||||
import { createDbAdapter, type CredentialLandingRow, type DbAdapter, type PublicUser } from '../db/index.js'
|
||||
import { createUserFs } from '../fs/provider.js'
|
||||
import { RemoteUserFs } from '../fs/remote-user-fs.js'
|
||||
import type { UserFs } from '../fs/user-fs.js'
|
||||
import { decrypt, deriveKey } from '../crypto.js'
|
||||
import { hashUid } from '../isolation.js'
|
||||
import { LocalSpawner } from '../supervisor/orchestrator.js'
|
||||
import { LeasedSpawner } from '../supervisor/leased-spawner.js'
|
||||
import { RemoteSpawner, type ClusterHost } from '../supervisor/remote-spawner.js'
|
||||
import { registerDshProxy } from '../supervisor/proxy.js'
|
||||
import type { Spawner } from '../supervisor/spawner.js'
|
||||
import {
|
||||
@@ -259,8 +262,147 @@ export async function buildServer(config: ServerConfig): Promise<FastifyInstance
|
||||
if (config.deployMode === 'k8s') {
|
||||
throw new Error('deployMode "k8s" is not supported by this build: only the single-machine backend ships')
|
||||
}
|
||||
const supervisor: Spawner = new LocalSpawner(config, resolveApiKey, resolveUid)
|
||||
const userFs = createUserFs(config)
|
||||
// cluster 模式(T08 S3/S4):实例在 worker 上,Manager 只投递操作 + 代理。
|
||||
// fail-loud:没配 agent 地址就直接报错,别等第一个用户点进来才发现。
|
||||
if (config.deployMode === 'cluster' && config.clusterAgentUrl === '') {
|
||||
throw new Error('deployMode=cluster requires DSHS_CLUSTER_AGENT_URL (e.g. http://127.0.0.1:9000)')
|
||||
}
|
||||
// cluster:RemoteSpawner(传输)+ LeasedSpawner(**归属租约**)——
|
||||
// 后者保证"能不能拉起先问归属",这是多机下防双写同一个 home 的承重件(设计 §3.2)。
|
||||
let leased: LeasedSpawner | undefined
|
||||
// ── 多 worker 的 host 目录(T08 S6)────────────────────────────────────
|
||||
// 由 `dsh_hosts` 派生并**随用随刷新**(TTL 30 s)⇒ **新增 worker 不必重启 Manager**。
|
||||
// 同时供三处使用:RemoteSpawner 的按 host 路由、LeasedSpawner 的 fence 目标、
|
||||
// 以及 `selectHost` 的容量准入 —— 都读**同一份**内存目录,避免三套各自漂移。
|
||||
const hostDirectory = new Map<string, ClusterHost>()
|
||||
hostDirectory.set(config.clusterHostId, {
|
||||
hostId: config.clusterHostId,
|
||||
agentUrl: config.clusterAgentUrl,
|
||||
token: config.clusterAgentToken,
|
||||
instanceHost: config.clusterInstanceHost,
|
||||
})
|
||||
const hostsProvider = async (): Promise<ClusterHost[]> => {
|
||||
for (const row of await db.listDshHosts()) {
|
||||
hostDirectory.set(row.id, {
|
||||
hostId: row.id,
|
||||
agentUrl: row.endpoint,
|
||||
token: row.agentToken,
|
||||
instanceHost: config.clusterInstanceHost,
|
||||
})
|
||||
}
|
||||
return [...hostDirectory.values()]
|
||||
}
|
||||
/**
|
||||
* 按用户归属解析 host(实例面与**文件面**共用这一份,避免两套路由漂移)。
|
||||
*
|
||||
* 为什么两处都要用:用户工作区在**那台 worker 的本地盘**;若文件面固定打一台 agent,
|
||||
* 就会出现「实例跑在 A、mkdir/上传写到 B」⇒ 实例看不到自己的文件、甚至 cwd 不存在而崩
|
||||
* (2026-09-15 生产切换暴露)。
|
||||
*/
|
||||
const hostIdForUser = async (userId: string): Promise<string | undefined> =>
|
||||
(await db.findUserInstance(userId, 'main'))?.hostId ?? undefined
|
||||
|
||||
/**
|
||||
* **文件面专用**路由:没有归属就**先选机并钉住**。
|
||||
*
|
||||
* 为什么不能直接用 hostIdForUser:新用户还没有归属,"写文件"和"launch"会各自选一次机,
|
||||
* 两次可能选到不同机器 ⇒「文件写到 A、实例起在 B」⇒ 实例看不到自己的文件(2026-09-15 实测)。
|
||||
* 首次触达工作区就把归属钉住,后续(含 launch)全走粘性 ⇒ 两面必然一致。
|
||||
*/
|
||||
const hostIdForFile = async (userId: string): Promise<string | undefined> => {
|
||||
const owned = await hostIdForUser(userId)
|
||||
if (owned !== undefined && owned !== null) return owned
|
||||
const chosen = (await selectHost(userId)) ?? config.clusterHostId
|
||||
if (chosen === '') return undefined
|
||||
await db.pinInstanceHost(userId, chosen)
|
||||
return chosen
|
||||
}
|
||||
/**
|
||||
* 选机:**① 粘性优先 ② 再按容量准入**。
|
||||
*
|
||||
* ⚠️ 顺序不能颠倒(2026-09-15 生产切换时补的缺口):用户工作区在**本地盘**、跟着机器走,
|
||||
* 把"已有历史数据的用户"调度到另一台 ⇒ 他打开实例看到**空工作区**。
|
||||
* ⇒ 有历史归属且那台还 `up` 就留在原地;只有**从未有过归属**(新用户)才按容量挑最空的。
|
||||
* `capacityMb <= 0` = 未声明(不设限);`-1` = 显式禁用承载。
|
||||
*/
|
||||
const reserveMb = Number(process.env.DSHS_CLUSTER_RESERVE_MB ?? '512')
|
||||
const selectHost = async (userId?: string): Promise<string | undefined> => {
|
||||
const rows = await db.listDshHosts()
|
||||
const eligible = rows.filter((h) => h.status === 'up' && h.capacityMb !== -1)
|
||||
if (userId !== undefined) {
|
||||
const owned = (await db.findUserInstance(userId, 'main'))?.hostId ?? null
|
||||
if (owned !== null && eligible.some((h) => h.id === owned)) return owned
|
||||
}
|
||||
const candidates = eligible.filter(
|
||||
(h) => h.capacityMb <= 0 || h.usedMb + reserveMb <= h.capacityMb,
|
||||
)
|
||||
if (candidates.length === 0) return undefined // 无候选 ⇒ 回退到配置里那台
|
||||
candidates.sort((a, b) => a.usedMb - b.usedMb)
|
||||
return candidates[0].id
|
||||
}
|
||||
const supervisor: Spawner =
|
||||
config.deployMode === 'cluster'
|
||||
? (leased = new LeasedSpawner(
|
||||
new RemoteSpawner({
|
||||
agentUrl: config.clusterAgentUrl,
|
||||
token: config.clusterAgentToken,
|
||||
instanceHost: config.clusterInstanceHost,
|
||||
defaultHostId: config.clusterHostId,
|
||||
hostsProvider,
|
||||
resolveApiKey,
|
||||
resolveUid,
|
||||
// 按 host 路由:每次操作都落到"该用户实例所在那台"(与文件面同一份)
|
||||
hostIdFor: hostIdForUser,
|
||||
}),
|
||||
db,
|
||||
{
|
||||
hostId: config.clusterHostId,
|
||||
agentUrl: config.clusterAgentUrl,
|
||||
agentToken: config.clusterAgentToken,
|
||||
capacityMb: Number(process.env.DSHS_CLUSTER_CAPACITY_MB ?? '0'),
|
||||
// 专用 Manager 部署设 DSHS_CLUSTER_REGISTER_SELF=0(见 LeasedSpawner 的注释)
|
||||
registerSelf: (process.env.DSHS_CLUSTER_REGISTER_SELF ?? '1') !== '0',
|
||||
ttlMs: Number(process.env.DSHS_CLUSTER_LEASE_TTL_MS ?? '30000'),
|
||||
renewMs: Number(process.env.DSHS_CLUSTER_LEASE_RENEW_MS ?? '10000'),
|
||||
selectHost,
|
||||
agentFor: (hostId: string) => {
|
||||
const h = hostDirectory.get(hostId)
|
||||
return h === undefined ? undefined : { agentUrl: h.agentUrl, token: h.token }
|
||||
},
|
||||
},
|
||||
))
|
||||
: new LocalSpawner(config, resolveApiKey, resolveUid)
|
||||
// 注册本机 + 起心跳(异步,不阻塞启动;心跳失败只影响该 worker 的状态位)
|
||||
if (leased !== undefined) {
|
||||
void leased.start().catch((err: unknown) => {
|
||||
console.error('[cluster] heartbeat/register failed to start:', err)
|
||||
})
|
||||
}
|
||||
const userFs = createUserFs(config, {
|
||||
hostIdFor: hostIdForFile,
|
||||
agentFor: (hostId: string) => {
|
||||
const h = hostDirectory.get(hostId)
|
||||
return h === undefined ? undefined : { agentUrl: h.agentUrl, token: h.token }
|
||||
},
|
||||
})
|
||||
// T08 S5:cluster 模式下**所有 worker 的 dataRoot 必须是同一绝对路径**(基线约定,
|
||||
// 设计 §14.3)。不一致会让 `resolvePath` 算出的"实例眼里的路径"与实际不符 ⇒
|
||||
// 文件面与 launch 的 folder 都会错。这里在启动时**报出来**,别等用户点进去才发现。
|
||||
if (userFs instanceof RemoteUserFs) {
|
||||
void userFs
|
||||
.probeWorkerRoot()
|
||||
.then((root) => {
|
||||
if (root !== undefined && root !== userFs.workerDataRoot) {
|
||||
console.error(
|
||||
`[cluster] worker dataRoot 与配置不一致:agent 报 ${root},本进程按 ${userFs.workerDataRoot} 计算路径。` +
|
||||
'请把 DSHS_CLUSTER_WORKER_DATA_ROOT 设为 worker 上的实际值(所有 worker 必须同路径)。',
|
||||
)
|
||||
}
|
||||
})
|
||||
.catch(() => {
|
||||
/* 探测失败不阻塞启动:会有心跳/调用失败暴露 */
|
||||
})
|
||||
}
|
||||
|
||||
const app = Fastify({
|
||||
logger: { level: config.logLevel },
|
||||
|
||||
@@ -0,0 +1,450 @@
|
||||
/**
|
||||
* Worker agent(T08 S3;设计 §11.2)。
|
||||
*
|
||||
* **它是什么**:Worker 上唯一的"被拨入口" —— 一个内部 HTTP 服务,把实例生命周期
|
||||
* 暴露给 Manager。**它不做归属决策**(谁托管谁是 Manager + PG 的事),只负责
|
||||
* "在这台机器上把实例起停好",并复用 **`LocalSpawner`**,因此 bwrap/uid/scope
|
||||
* 隔离、内存配额推导、崩溃退避与熔断、插件探活这些**本地语义全部原样保留**
|
||||
* (这是本方案相对 k8s 路线最大的成本优势)。
|
||||
*
|
||||
* 四条协议纪律(设计 §11.3):
|
||||
* 1. **单向拨入**:Worker 不反向连 Manager、不写控制面数据(自己可以有库,见下);
|
||||
* 2. **幂等键**:每个变更请求带 `operationId`,重复请求**回放上次结果**
|
||||
* (否则 Manager 超时重试会起两个实例);
|
||||
* 3. **最小接口**:只接受白名单动作,参数受限(folder 由 Manager 解析、patch 有长度上限)
|
||||
* —— agent 若能被当任意命令执行器,Worker 沦陷 = 全集群沦陷;
|
||||
* 4. **self-fencing**:`POST /fence {userId, epoch}` —— 本地记录的 epoch 落后于
|
||||
* Manager 下发的值 ⇒ **主动停掉该实例**(防双写的最后一道防线)。
|
||||
*
|
||||
* @module dshs/worker/agent
|
||||
*/
|
||||
import { timingSafeEqual } from 'node:crypto'
|
||||
import Fastify, { type FastifyInstance, type FastifyReply } from 'fastify'
|
||||
import type { ServerConfig } from '../config.js'
|
||||
import { LocalUserFs } from '../fs/local-user-fs.js'
|
||||
import { userRoot } from '../fs/workspace.js'
|
||||
import { isUserFsErrorCode, UserFsError } from '../fs/user-fs.js'
|
||||
import { hashUid } from '../isolation.js'
|
||||
import { LocalSpawner } from '../supervisor/orchestrator.js'
|
||||
import { SshTunnel } from './tunnel.js'
|
||||
import type { Instance } from '../supervisor/spawner.js'
|
||||
|
||||
/** 绑定的头部名(Manager/agent 双方约定)。 */
|
||||
export const AGENT_TOKEN_HEADER = 'x-dsh-agent-token'
|
||||
|
||||
export interface WorkerAgentOptions {
|
||||
/** 本机在 `dsh_hosts.id` 里的标识。 */
|
||||
hostId: string
|
||||
/** 共享密钥(仅内网 + nft 白名单;本版是 bearer 式比较,HMAC/防重放留待后续)。 */
|
||||
token: string
|
||||
/** 监听端口。 */
|
||||
port: number
|
||||
/** 绑定地址(默认 `0.0.0.0`,靠 nft 只放行 Manager 网段)。 */
|
||||
host?: string
|
||||
/** 返回给 Manager 做代理的地址(同机 1a 用 `127.0.0.1`;跨机时填内网 IP)。 */
|
||||
instanceHost?: string
|
||||
/** 日志级别。 */
|
||||
logLevel?: string
|
||||
/**
|
||||
* **反向隧道**(跨机演练):Worker 主动拨 Manager,形如 `[email protected]:32022`。
|
||||
* 不设则完全关闭(同机/单机形态零影响)。见 `tunnel.ts` 头注释。
|
||||
*/
|
||||
tunnelTarget?: string
|
||||
/** 隧道私钥(默认 `~/.ssh/tunnel_ed25519`)。 */
|
||||
tunnelIdentity?: string
|
||||
/** ControlMaster socket(默认 `/tmp/dshs-tunnel-<hostId>.sock`)。 */
|
||||
tunnelControlPath?: string
|
||||
}
|
||||
|
||||
/** 变更类请求的幂等缓存条数上限(超出后丢最旧的 —— 只是省重试,不是审计)。 */
|
||||
const OP_CACHE_MAX = 512
|
||||
/** patch 内容长度上限(防把 agent 当大对象存储)。 */
|
||||
const MAX_PATCH_BYTES = 256 * 1024
|
||||
|
||||
interface OpCache {
|
||||
order: string[]
|
||||
results: Map<string, unknown>
|
||||
}
|
||||
|
||||
/**
|
||||
* 组装 agent。**复用 `LocalSpawner`**(隔离/配额/退避/熔断/探活全部原样保留)。
|
||||
*
|
||||
* ⚠️ **边界要读准**(2026-09-15 用户纠正):本 agent **不写控制面数据**(尤其归属/租约 ——
|
||||
* 双写就是脑裂),所以 `apiKey` 与 `uid` **不由本机查控制面库**,而是 Manager 在
|
||||
* `POST /launch` 时随请求投递(与 k8s 用 per-user Secret 同一思路),只存内存。
|
||||
* 但这**不等于"Worker 不许有数据库"**:插件的 per-user 数据(如 `home/.dsh/mcn-plugin.db`)
|
||||
* 属于**实例业务数据**,由实例自己读写、跟着 home 走;Worker 也可以有自己的运维库。
|
||||
* 完整判据见设计 §1.3「数据分层」。
|
||||
*/
|
||||
export function buildWorkerAgent(
|
||||
config: ServerConfig,
|
||||
options: WorkerAgentOptions,
|
||||
): { app: FastifyInstance; spawner: LocalSpawner; stop: () => Promise<void> } {
|
||||
/** launch 时投递、仅存内存的凭据与 uid(Worker 不连 DB)。 */
|
||||
const apiKeys = new Map<string, string>()
|
||||
const uids = new Map<string, number>()
|
||||
const spawner = new LocalSpawner(
|
||||
config,
|
||||
async (userId: string) => apiKeys.get(userId) ?? null,
|
||||
async (userId: string) => uids.get(userId) ?? hashUid(userId, config.baseUid),
|
||||
)
|
||||
/**
|
||||
* 文件面(T08 S5):**复用同一个 `LocalUserFs`** —— 用户卷本来就在本机,
|
||||
* 所以"跨机文件面"= 把这个实现经 HTTP 暴露出去,而不是重新实现一套路径语义。
|
||||
*/
|
||||
const userFs = new LocalUserFs((userId: string) => userRoot(config.dataRoot, userId))
|
||||
|
||||
/**
|
||||
* 反向隧道(可选)。静态转发 = **agent 自身端口** + `DSHS_TUNNEL_STATIC_PORTS`(如控制面 PG);
|
||||
* 实例端口在 launch/stop 时动态加减,并在 `/healthz`(Manager 的心跳)里**对账自愈**。
|
||||
*/
|
||||
const tunnelTarget = options.tunnelTarget ?? process.env.DSHS_TUNNEL_TARGET ?? ''
|
||||
const staticPorts = [
|
||||
options.port,
|
||||
...(process.env.DSHS_TUNNEL_STATIC_PORTS ?? '')
|
||||
.split(',')
|
||||
.map((v) => Number(v.trim()))
|
||||
.filter((v) => Number.isInteger(v) && v > 0),
|
||||
]
|
||||
const tunnel =
|
||||
tunnelTarget === ''
|
||||
? undefined
|
||||
: new SshTunnel({
|
||||
target: tunnelTarget,
|
||||
identity:
|
||||
options.tunnelIdentity ??
|
||||
process.env.DSHS_TUNNEL_IDENTITY ??
|
||||
`${process.env.HOME ?? '/root'}/.ssh/tunnel_ed25519`,
|
||||
controlPath: options.tunnelControlPath ?? `/tmp/dshs-tunnel-${options.hostId}.sock`,
|
||||
staticPorts,
|
||||
})
|
||||
let tunnelReady = tunnel === undefined
|
||||
|
||||
/**
|
||||
* **隧道自愈**:master 失联就重建(重建会自动补回 staticPorts 的静态转发)。
|
||||
*
|
||||
* ⚠️ 为什么不能只在 `/healthz` 里做(2026-09-15 想清楚的一个死角):`/healthz` 是**经隧道**
|
||||
* 才打得进来的 —— 隧道一断,Manager 的心跳就进不来,自愈**永远不会被触发**(自己把自己锁死)。
|
||||
* ⇒ 必须由 **agent 本地定时器**驱动(下面 20s 一跳),`/healthz` 里再顺手做一次。
|
||||
*/
|
||||
const healTunnel = async (): Promise<void> => {
|
||||
if (tunnel === undefined) return
|
||||
if (tunnelReady && (await tunnel.isMasterAlive())) return
|
||||
tunnelReady = false
|
||||
try {
|
||||
await tunnel.ensureMaster()
|
||||
tunnelReady = true
|
||||
console.error('[tunnel] master 失联 → 已重建(含静态转发)')
|
||||
} catch (err) {
|
||||
console.error('[tunnel] 重建失败,下轮再试:', err instanceof Error ? err.message : err)
|
||||
}
|
||||
}
|
||||
|
||||
/** 把活着的实例端口补齐、把已消失的撤掉(崩溃退出也走这里收敛,不必逐个挂 exit 钩子)。 */
|
||||
const reconcileTunnel = async (): Promise<void> => {
|
||||
if (tunnel === undefined || !tunnelReady) return
|
||||
const live = new Set((await spawner.listUserInstances()).map((i) => i.port).filter((p): p is number => p !== undefined))
|
||||
for (const port of live) await tunnel.forward(port)
|
||||
for (const port of tunnel.ports) {
|
||||
if (!live.has(port) && !staticPorts.includes(port)) await tunnel.cancel(port)
|
||||
}
|
||||
}
|
||||
|
||||
let tunnelTimer: NodeJS.Timeout | undefined
|
||||
if (tunnel !== undefined) {
|
||||
void tunnel
|
||||
.ensureMaster()
|
||||
.then(() => {
|
||||
tunnelReady = true
|
||||
})
|
||||
.catch((err: unknown) => {
|
||||
console.error('[tunnel] 建立失败(跨机代理将不可用,本机功能不受影响):', err instanceof Error ? err.message : err)
|
||||
})
|
||||
tunnelTimer = setInterval(() => {
|
||||
void healTunnel().then(reconcileTunnel)
|
||||
}, 20_000)
|
||||
tunnelTimer.unref?.()
|
||||
}
|
||||
const app = Fastify({ logger: { level: options.logLevel ?? 'info' }, bodyLimit: MAX_PATCH_BYTES + 4096 })
|
||||
const cache: OpCache = { order: [], results: new Map() }
|
||||
/** agent 侧记住的 epoch(self-fencing 判据)。 */
|
||||
const epochs = new Map<string, number>()
|
||||
|
||||
const remember = (op: string, value: unknown): void => {
|
||||
if (cache.results.has(op)) return
|
||||
cache.results.set(op, value)
|
||||
cache.order.push(op)
|
||||
while (cache.order.length > OP_CACHE_MAX) {
|
||||
const oldest = cache.order.shift()
|
||||
if (oldest !== undefined) cache.results.delete(oldest)
|
||||
}
|
||||
}
|
||||
|
||||
app.addHook('onRequest', async (request, reply) => {
|
||||
if (request.url === '/healthz') return // 存活探测不带凭据
|
||||
const given = request.headers[AGENT_TOKEN_HEADER]
|
||||
const expected = options.token
|
||||
const a = Buffer.from(typeof given === 'string' ? given : '')
|
||||
const b = Buffer.from(expected)
|
||||
if (a.length !== b.length || !timingSafeEqual(a, b)) {
|
||||
await reply.code(401).send({ error: 'unauthorized' })
|
||||
}
|
||||
})
|
||||
|
||||
app.get('/healthz', async () => {
|
||||
const instances = await spawner.listUserInstances()
|
||||
// 心跳里顺手自愈 + 对账(主驱动是本地定时器,见 healTunnel 的注释)
|
||||
await healTunnel()
|
||||
await reconcileTunnel()
|
||||
return {
|
||||
ok: true,
|
||||
hostId: options.hostId,
|
||||
instances: instances.length,
|
||||
tunnel: tunnel === undefined ? null : { ready: tunnelReady, ports: tunnel.ports },
|
||||
}
|
||||
})
|
||||
|
||||
/** 对账用:**一次拿回整机**(设计 §11.6,替代逐用户查询)。 */
|
||||
app.get('/instances', async () => ({ instances: await spawner.listUserInstances() }))
|
||||
|
||||
app.post('/launch', async (request, reply) => {
|
||||
const body = request.body as {
|
||||
userId?: string
|
||||
folder?: string
|
||||
patch?: string
|
||||
epoch?: number
|
||||
operationId?: string
|
||||
apiKey?: string | null
|
||||
uid?: number
|
||||
}
|
||||
if (body.userId === undefined || body.operationId === undefined) {
|
||||
return reply.code(400).send({ error: 'userId and operationId are required' })
|
||||
}
|
||||
const cached = cache.results.get(body.operationId)
|
||||
if (cached !== undefined) return cached // 幂等回放
|
||||
// 先落凭据/uid/**epoch**(重放路径也安全:同值覆盖)。
|
||||
// ⚠️ epoch 记录的是「Manager 的意图」,因此必须在**尝试 spawn 之前**落 ——
|
||||
// "实例本来就在跑"(AlreadyRunningError 分支)时也要记,否则 `/fence` 拿不到
|
||||
// 我的 epoch,self-fencing 就永远不触发(2026-09-15 T08 S3 实测踩到)。
|
||||
if (body.apiKey !== undefined && body.apiKey !== null) apiKeys.set(body.userId, body.apiKey)
|
||||
if (body.uid !== undefined) uids.set(body.userId, body.uid)
|
||||
if (body.epoch !== undefined) epochs.set(body.userId, body.epoch)
|
||||
try {
|
||||
const instance = await spawner.launch(body.userId, body.folder ?? '', body.patch)
|
||||
// 跨机:把该实例端口经隧道打到 Manager 侧(失败不阻断 —— 本机仍可用)
|
||||
if (tunnel !== undefined && tunnelReady && instance.port !== undefined) await tunnel.forward(instance.port)
|
||||
const payload = { instance: { ...instance, launchToken: spawner.launchTokenOf(body.userId) } }
|
||||
remember(body.operationId, payload)
|
||||
return payload
|
||||
} catch (err) {
|
||||
// 已在跑:**返回现有实例**而不是报错 —— 这让重试天然安全(与 AlreadyRunningError 语义对齐)。
|
||||
const msg = err instanceof Error ? err.message : String(err)
|
||||
if (/already has a running/i.test(msg)) {
|
||||
const currents = await spawner.listUserInstances()
|
||||
const found = currents.find((i) => i.userId === body.userId)
|
||||
if (found !== undefined) {
|
||||
const payload = { instance: { ...found, launchToken: spawner.launchTokenOf(body.userId) }, note: 'already-running' }
|
||||
remember(body.operationId, payload)
|
||||
return payload
|
||||
}
|
||||
}
|
||||
return reply.code(500).send({ error: msg })
|
||||
}
|
||||
})
|
||||
|
||||
app.post('/stop', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; operationId?: string }
|
||||
if (body.userId === undefined || body.operationId === undefined) {
|
||||
return reply.code(400).send({ error: 'userId and operationId are required' })
|
||||
}
|
||||
const cached = cache.results.get(body.operationId)
|
||||
if (cached !== undefined) return cached
|
||||
const before = await spawner.status(body.userId)
|
||||
await spawner.stop(body.userId)
|
||||
if (tunnel !== undefined && tunnelReady && before.main?.port !== undefined) await tunnel.cancel(before.main.port)
|
||||
epochs.delete(body.userId)
|
||||
apiKeys.delete(body.userId) // 凭据只该活在实例生命周期内
|
||||
const payload = { ok: true }
|
||||
remember(body.operationId, payload)
|
||||
return payload
|
||||
})
|
||||
|
||||
app.get('/status/:userId', async (request, reply) => {
|
||||
const { userId } = request.params as { userId: string }
|
||||
const status = await spawner.status(userId)
|
||||
return reply.send({ userId, main: status.main ?? null })
|
||||
})
|
||||
|
||||
/** 代理目标(Manager 用)。未运行时返回 `{ running: false }`。 */
|
||||
app.get('/endpoint/:userId', async (request) => {
|
||||
const { userId } = request.params as { userId: string }
|
||||
const endpoint = await spawner.endpointFor(userId)
|
||||
return endpoint === undefined
|
||||
? { running: false }
|
||||
: { running: true, host: options.instanceHost ?? '127.0.0.1', port: endpoint.port }
|
||||
})
|
||||
|
||||
/** self-fencing:我持有的 epoch 落后于 Manager 下发的值 ⇒ **自杀**。 */
|
||||
app.post('/fence', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; epoch?: number }
|
||||
if (body.userId === undefined || body.epoch === undefined) {
|
||||
return reply.code(400).send({ error: 'userId and epoch are required' })
|
||||
}
|
||||
const mine = epochs.get(body.userId)
|
||||
if (mine === undefined || mine >= body.epoch) return { fenced: false, mine: mine ?? null }
|
||||
await spawner.stop(body.userId)
|
||||
epochs.delete(body.userId)
|
||||
return { fenced: true, mine }
|
||||
})
|
||||
|
||||
app.post('/restart-probe/:userId', async (request) => {
|
||||
const { userId } = request.params as { userId: string }
|
||||
return spawner.restartAndProbe(userId)
|
||||
})
|
||||
|
||||
/**
|
||||
* 活动信号转发(Manager 代理到用户流量时调用)。
|
||||
* 为什么要转发:idle-reap 是**本地语义**(`LocalSpawner` 的 `lastActive` + TTL/LRU),
|
||||
* 不转发的话 worker 会以为实例一直没人用、把它回收掉(档案 08)。
|
||||
*/
|
||||
app.post('/touch/:userId', async (request) => {
|
||||
const { userId } = request.params as { userId: string }
|
||||
spawner.touch(userId)
|
||||
return { ok: true }
|
||||
})
|
||||
|
||||
app.post('/watchdog/:userId', async (request) => {
|
||||
const { userId } = request.params as { userId: string }
|
||||
return { instance: (await spawner.spawnWatchdog(userId)) ?? null }
|
||||
})
|
||||
|
||||
// ── 文件面(T08 S5;供 Manager 的 RemoteUserFs 调用)──────────────────────
|
||||
// 请求体/响应都是**工作区相对路径 + base64**,与 `UserFs` 的语义一一对应;
|
||||
// 失败时回 `{error: code}` 并把 `UserFsError.code` 映射成对应 HTTP 状态
|
||||
// —— 这正是 `user-fs.ts` 里那个 seam 设计的用法(路由按 code 回给前端)。
|
||||
|
||||
/** 统一包装:把 `UserFsError` 还原成 wire 形态(其余错误 → 500)。 */
|
||||
const fsCall = async <T>(reply: FastifyReply, fn: () => Promise<T>): Promise<T | undefined> => {
|
||||
try {
|
||||
return await fn()
|
||||
} catch (err) {
|
||||
if (err instanceof UserFsError) {
|
||||
await reply.code(err.status).send({ error: err.code })
|
||||
return undefined
|
||||
}
|
||||
const msg = err instanceof Error ? err.message : String(err)
|
||||
await reply.code(500).send({ error: 'internal', detail: msg.slice(0, 200) })
|
||||
return undefined
|
||||
}
|
||||
}
|
||||
|
||||
app.post('/fs/init', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; uid?: number }
|
||||
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
|
||||
return (await fsCall(reply, async () => {
|
||||
await userFs.initUserRoot(body.userId as string, body.uid)
|
||||
return { ok: true }
|
||||
})) ?? reply
|
||||
})
|
||||
|
||||
app.post('/fs/list', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; relPath?: string }
|
||||
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
|
||||
return (await fsCall(reply, () => userFs.listDir(body.userId as string, body.relPath ?? ''))) ?? reply
|
||||
})
|
||||
|
||||
app.post('/fs/mkdir', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; relPath?: string }
|
||||
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
|
||||
return (await fsCall(reply, async () => {
|
||||
await userFs.mkdir(body.userId as string, body.relPath ?? '')
|
||||
return { ok: true }
|
||||
})) ?? reply
|
||||
})
|
||||
|
||||
app.post('/fs/create', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; relPath?: string; name?: string; type?: 'file' | 'dir' }
|
||||
if (body.userId === undefined || body.name === undefined) {
|
||||
return reply.code(400).send({ error: 'userId and name are required' })
|
||||
}
|
||||
// ⚠️ 必须包成对象:`createEntry` 返回的是字符串(净化后的文件名),直接 return 会被
|
||||
// Fastify 当 text/plain 发出,而调用方(RemoteUserFs)按 JSON 解析 ⇒ 静默 500。
|
||||
const created = await fsCall(reply, () =>
|
||||
userFs.createEntry(body.userId as string, body.relPath ?? '', body.name as string, body.type ?? 'file'),
|
||||
)
|
||||
return created === undefined ? reply : { name: created }
|
||||
})
|
||||
|
||||
app.post('/fs/upload', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; relPath?: string; name?: string; dataBase64?: string }
|
||||
if (body.userId === undefined || body.name === undefined || body.dataBase64 === undefined) {
|
||||
return reply.code(400).send({ error: 'userId, name and dataBase64 are required' })
|
||||
}
|
||||
// 同上:`upload` 返回的是净化后的文件名,必须包成对象。
|
||||
const uploaded = await fsCall(reply, () =>
|
||||
userFs.upload(
|
||||
body.userId as string,
|
||||
body.relPath ?? '',
|
||||
body.name as string,
|
||||
Buffer.from(body.dataBase64 as string, 'base64'),
|
||||
),
|
||||
)
|
||||
return uploaded === undefined ? reply : { name: uploaded }
|
||||
})
|
||||
|
||||
app.post('/fs/read', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; relPath?: string; maxBytes?: number }
|
||||
if (body.userId === undefined || body.relPath === undefined) {
|
||||
return reply.code(400).send({ error: 'userId and relPath are required' })
|
||||
}
|
||||
const out = await fsCall(reply, () => userFs.readFile(body.userId as string, body.relPath as string, body.maxBytes))
|
||||
if (out === undefined) return reply
|
||||
return { name: out.name, dataBase64: out.data.toString('base64') }
|
||||
})
|
||||
|
||||
app.post('/fs/isdir', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; relPath?: string }
|
||||
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
|
||||
const out = await fsCall(reply, () => userFs.isDirectory(body.userId as string, body.relPath ?? ''))
|
||||
return out === undefined ? reply : { isDirectory: out }
|
||||
})
|
||||
|
||||
app.post('/fs/plugins', async (request, reply) => {
|
||||
const body = request.body as { userId?: string }
|
||||
if (body.userId === undefined) return reply.code(400).send({ error: 'userId is required' })
|
||||
return (await fsCall(reply, () => userFs.listInstalledPlugins(body.userId as string))) ?? reply
|
||||
})
|
||||
|
||||
app.post('/fs/handoff', async (request, reply) => {
|
||||
const body = request.body as { userId?: string; content?: string }
|
||||
if (body.userId === undefined || body.content === undefined) {
|
||||
return reply.code(400).send({ error: 'userId and content are required' })
|
||||
}
|
||||
return (await fsCall(reply, async () => {
|
||||
await userFs.writeHandoff(body.userId as string, body.content as string)
|
||||
return { ok: true }
|
||||
})) ?? reply
|
||||
})
|
||||
|
||||
/** 本机 dataRoot(Manager 的 RemoteUserFs 用它做 `resolvePath` 的路径数学)。 */
|
||||
app.get('/fs/root', async () => ({ dataRoot: config.dataRoot }))
|
||||
|
||||
app.get('/', async () => ({ agent: 'dshs-worker', hostId: options.hostId }))
|
||||
|
||||
return {
|
||||
app,
|
||||
spawner,
|
||||
/**
|
||||
* 停机:**先收实例、再关 HTTP**。
|
||||
* 为什么必须收:实例是 worker 自己的子进程,停机不收就变孤儿(占用端口与内存);
|
||||
* 而且孤儿会继承 stdout ⇒ 调用方的管道永不关闭(2026-09-15 实测:verify 脚本挂死)。
|
||||
* 归属与租约由 Manager 侧处理(worker 不写控制面数据),所以这里只停进程。
|
||||
*/
|
||||
stop: async (): Promise<void> => {
|
||||
if (tunnelTimer !== undefined) clearInterval(tunnelTimer)
|
||||
await app.close()
|
||||
await spawner.teardown()
|
||||
await tunnel?.close()
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
/**
|
||||
* Worker 侧的**反向隧道管理器**(T08 跨机演练)。
|
||||
*
|
||||
* 为什么需要它:Manager 要连 Worker 上的两样东西 —— **agent 端口**与**每个实例的端口**
|
||||
* (实例只监听 `127.0.0.1`,这是 portGuard 的设计前提)。而 Worker 公网入方向通常被
|
||||
* **云安全组**挡住(实测:106 的 19100 从 47 与本机都连不上),放通只能在控制台点。
|
||||
*
|
||||
* 绕法:**让 Worker 主动拨 Manager**,用 SSH 反向转发把两边的 `127.0.0.1:<port>` 接起来。
|
||||
* 好处(实测):
|
||||
* · **两端都不用新开端口** —— 只用已开放的 SSH 端口(47 是 32022);
|
||||
* · 链路是加密的,且 Manager 侧落在 loopback(`GatewayPorts no` 默认)⇒ 不对外暴露;
|
||||
* · 实例端口是**动态**的(`findFreePort()`)⇒ 用 **ControlMaster + `ssh -O forward/cancel`**
|
||||
* 在**同一条长连接**上加/减转发,不必为每个端口重开连接。
|
||||
*
|
||||
* ⚠️ 定位:这是**演练级**传输(生产长期方案见设计 §2.3:受控网段白名单或隧道服务)。
|
||||
* ⚠️ 默认**关闭**:只有设了 `DSHS_TUNNEL_TARGET` 才启用 ⇒ 对同机/单机形态零影响。
|
||||
*
|
||||
* @module dshs/worker/tunnel
|
||||
*/
|
||||
import { execFile } from 'node:child_process'
|
||||
import { existsSync, unlinkSync } from 'node:fs'
|
||||
import { promisify } from 'node:util'
|
||||
|
||||
const run = promisify(execFile)
|
||||
|
||||
export interface TunnelOptions {
|
||||
/** 拨入目标,形如 `[email protected]:32022`。 */
|
||||
target: string
|
||||
/** 私钥路径(建议专用、且在 Manager 侧用 `restrict,port-forwarding` 限权)。 */
|
||||
identity: string
|
||||
/** ControlMaster socket 路径(同一路径复用同一条连接)。 */
|
||||
controlPath: string
|
||||
/** 启动时就转发的端口(agent 自身;还可带控制面 PG 等)。 */
|
||||
staticPorts?: number[]
|
||||
/** `ssh` 可执行文件路径。 */
|
||||
sshBin?: string
|
||||
}
|
||||
|
||||
export class SshTunnel {
|
||||
/** 内部一律用**已补默认值**的具体类型(否则 `sshBin` 会是 `string | undefined`)。 */
|
||||
private readonly opts: {
|
||||
target: string
|
||||
identity: string
|
||||
controlPath: string
|
||||
staticPorts: number[]
|
||||
sshBin: string
|
||||
}
|
||||
private readonly forwarded = new Set<number>()
|
||||
private readonly hostPart: string
|
||||
private readonly portPart: number | undefined
|
||||
|
||||
constructor(options: TunnelOptions) {
|
||||
// `user@host:port` 里的 port 是 **SSH 端口**(不是转发的端口)—— 47 上用 32022,
|
||||
// 必须经 `-p` 传,否则会去连 22 而失败。
|
||||
const [hostPart, portPart] = options.target.split(':')
|
||||
this.hostPart = hostPart
|
||||
this.portPart = portPart === undefined ? undefined : Number(portPart)
|
||||
this.opts = {
|
||||
target: options.target,
|
||||
identity: options.identity,
|
||||
controlPath: options.controlPath,
|
||||
staticPorts: options.staticPorts ?? [],
|
||||
sshBin: options.sshBin ?? '/usr/bin/ssh',
|
||||
}
|
||||
}
|
||||
|
||||
/** 所有 ssh 调用的公共参数(`-p` 只在目标里显式给了端口时才加)。 */
|
||||
private baseArgs(): string[] {
|
||||
return this.portPart === undefined ? [] : ['-p', String(this.portPart)]
|
||||
}
|
||||
|
||||
/** 当前已转发的端口(诊断用)。 */
|
||||
get ports(): number[] {
|
||||
return [...this.forwarded]
|
||||
}
|
||||
|
||||
/**
|
||||
* 建立(或复用)ControlMaster 长连接,并把 `staticPorts` 转发上去。
|
||||
* 幂等:socket 已存在且 master 还活着就直接返回。
|
||||
*/
|
||||
async ensureMaster(): Promise<void> {
|
||||
if (existsSync(this.opts.controlPath)) {
|
||||
try {
|
||||
await run(this.opts.sshBin, [...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'check', this.hostPart])
|
||||
// master 活着 ⇒ 只需补齐静态转发
|
||||
for (const port of this.opts.staticPorts ?? []) await this.forward(port)
|
||||
return
|
||||
} catch {
|
||||
try {
|
||||
unlinkSync(this.opts.controlPath) // 僵尸 socket:清掉重建
|
||||
} catch {
|
||||
/* 无所谓 */
|
||||
}
|
||||
}
|
||||
}
|
||||
const args = [
|
||||
'-M',
|
||||
'-N',
|
||||
'-f',
|
||||
...this.baseArgs(),
|
||||
'-S',
|
||||
this.opts.controlPath,
|
||||
'-i',
|
||||
this.opts.identity,
|
||||
'-o',
|
||||
'BatchMode=yes',
|
||||
'-o',
|
||||
'StrictHostKeyChecking=accept-new',
|
||||
'-o',
|
||||
'ExitOnForwardFailure=yes',
|
||||
'-o',
|
||||
'ServerAliveInterval=15',
|
||||
'-o',
|
||||
'ServerAliveCountMax=4',
|
||||
]
|
||||
for (const port of this.opts.staticPorts ?? []) args.push('-R', `${port}:127.0.0.1:${port}`)
|
||||
args.push(this.hostPart)
|
||||
await run(this.opts.sshBin, args, { timeout: 20_000 })
|
||||
for (const port of this.opts.staticPorts ?? []) this.forwarded.add(port)
|
||||
}
|
||||
|
||||
/**
|
||||
* master 是否还活着(`ssh -O check`)。
|
||||
*
|
||||
* 为什么需要:**对端 SSH 重启/断链后,转发会全部消失,而本地 `forwarded` 集合并不知情**
|
||||
* ⇒ 若只看本地状态,会以为"隧道还好",实际 Manager 已经连不上这台 Worker
|
||||
* (2026-09-15 收口"以跑通为目的"时补)。
|
||||
*/
|
||||
async isMasterAlive(): Promise<boolean> {
|
||||
try {
|
||||
await run(this.opts.sshBin, [...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'check', this.hostPart], {
|
||||
timeout: 8_000,
|
||||
})
|
||||
return true
|
||||
} catch {
|
||||
// check 失败 ⇒ master 不在了;顺手清掉本地记账,避免"以为还转着"
|
||||
this.forwarded.clear()
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 动态加一条反向转发(实例起来时调用)。
|
||||
* 用**同一个端口号**:实例在 Worker 上是 `127.0.0.1:<port>`,反向转发落到 Manager 的
|
||||
* `127.0.0.1:<port>` ⇒ Manager 侧无需端口映射表,`endpointFor` 直接回 `127.0.0.1`。
|
||||
*/
|
||||
async forward(port: number): Promise<boolean> {
|
||||
if (this.forwarded.has(port)) return true
|
||||
try {
|
||||
await run(
|
||||
this.opts.sshBin,
|
||||
[...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'forward', '-R', `${port}:127.0.0.1:${port}`, this.hostPart],
|
||||
{ timeout: 10_000 },
|
||||
)
|
||||
this.forwarded.add(port)
|
||||
return true
|
||||
} catch {
|
||||
return false // 失败不抛:实例本身仍在本机可用,只是跨机代理这跳不可用
|
||||
}
|
||||
}
|
||||
|
||||
/** 撤销一条转发(实例停止/退出时调用)。 */
|
||||
async cancel(port: number): Promise<void> {
|
||||
if (!this.forwarded.has(port)) return
|
||||
try {
|
||||
await run(
|
||||
this.opts.sshBin,
|
||||
[...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'cancel', '-R', `${port}:127.0.0.1:${port}`, this.hostPart],
|
||||
{ timeout: 10_000 },
|
||||
)
|
||||
} catch {
|
||||
/* 连接已断也一样算撤销 */
|
||||
}
|
||||
this.forwarded.delete(port)
|
||||
}
|
||||
|
||||
/** 关闭 master(进程退出时)。 */
|
||||
async close(): Promise<void> {
|
||||
try {
|
||||
await run(this.opts.sshBin, [...this.baseArgs(), '-S', this.opts.controlPath, '-O', 'exit', this.hostPart], {
|
||||
timeout: 10_000,
|
||||
})
|
||||
} catch {
|
||||
/* 已退出 */
|
||||
}
|
||||
this.forwarded.clear()
|
||||
}
|
||||
|
||||
/** 目标 SSH 端口(`root@h:32022` → 32022)。 */
|
||||
get targetPort(): number | undefined {
|
||||
return this.portPart
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,183 @@
|
||||
/**
|
||||
* T08 S2 · 实例归属租约单测。
|
||||
*
|
||||
* 刻意**不 mock 时钟**:走真实 SQL(SqliteAdapter(':memory:'))并用**极短 TTL** 制造过期,
|
||||
* 这样测到的是"SQL 的原子抢占真的成立",而不是"我的假时钟算对了"。
|
||||
*
|
||||
* 两个后端都跑(同 T08 S1 的做法):默认 SQLite(内存库,每用例一份);
|
||||
* 设 `LEASETEST_PG_URL=postgres://…` 时改跑 PG —— 用来验证两套实现语义一致。
|
||||
* 运行:node --test test/lease.test.mjs
|
||||
* LEASETEST_PG_URL=postgres://dshs:[email protected]:15432/dshs_smoke node --test test/lease.test.mjs
|
||||
*/
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
import pg from 'pg'
|
||||
|
||||
import { SqliteAdapter } from '../lib/db/sqlite.js'
|
||||
import { openPgAdapter } from '../lib/db/pg.js'
|
||||
import { InstanceLease, stillHolder, DEFAULT_LEASE_TTL_MS } from '../lib/supervisor/lease.js'
|
||||
|
||||
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms))
|
||||
const PG_URL = process.env.LEASETEST_PG_URL
|
||||
|
||||
/** PG 侧:每个用例前清空三张表(该库专供本测试,清空是安全的)。 */
|
||||
async function resetPg() {
|
||||
const client = new pg.Client({ connectionString: PG_URL })
|
||||
await client.connect()
|
||||
await client.query('DELETE FROM dsh_instances')
|
||||
await client.query('DELETE FROM dsh_hosts')
|
||||
await client.query('DELETE FROM users')
|
||||
await client.end()
|
||||
}
|
||||
|
||||
/** 每个用例一个独立后端(SQLite = 新内存库;PG = 清空后的专用库)。 */
|
||||
async function freshDb() {
|
||||
const db = PG_URL === undefined ? new SqliteAdapter(':memory:', 100000) : await openPgAdapter(PG_URL, 100000)
|
||||
if (PG_URL !== undefined) await resetPg()
|
||||
await db.createUser({
|
||||
id: 'u1',
|
||||
username: 'alice',
|
||||
passHash: 'x',
|
||||
role: 'active',
|
||||
homeDir: '/tmp/u1',
|
||||
})
|
||||
return db
|
||||
}
|
||||
|
||||
/** 短 TTL 的租约(ttl=60ms > 2×20ms,满足不变量)。 */
|
||||
function shortLease(db, hostId) {
|
||||
return new InstanceLease(db, hostId, { ttlMs: 60, renewMs: 20 })
|
||||
}
|
||||
|
||||
test('lease: 默认时序满足 ttl > 2×renew 不变量', () => {
|
||||
assert.ok(DEFAULT_LEASE_TTL_MS > 2 * 10_000, '默认 30s TTL 必须 > 2×10s 续租')
|
||||
})
|
||||
|
||||
test('lease: 违反不变量时构造即抛(fail-loud,防抖动误判)', async () => {
|
||||
const db = await freshDb()
|
||||
assert.throws(() => new InstanceLease(db, 'w-a', { ttlMs: 100, renewMs: 60 }), /ttlMs/)
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: 首次抢占成功,epoch 从 1 开始', async () => {
|
||||
const db = await freshDb()
|
||||
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
|
||||
const r = await a.acquire('u1')
|
||||
assert.equal(r.ok, true)
|
||||
assert.equal(r.epoch, 1)
|
||||
assert.ok(r.leaseUntil > Date.now(), '租约应在未来')
|
||||
assert.deepEqual(a.holdings().get('u1'), { epoch: 1, hostId: 'w-a' })
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: 未过期时他人抢占失败 —— 单写者保证', async () => {
|
||||
const db = await freshDb()
|
||||
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
|
||||
const b = new InstanceLease(db, 'w-b', { ttlMs: 5000, renewMs: 1000 })
|
||||
assert.equal((await a.acquire('u1')).ok, true)
|
||||
|
||||
const r = await b.acquire('u1')
|
||||
assert.equal(r.ok, false, '有人在管 ⇒ 必须退让')
|
||||
assert.equal(r.holder, 'w-a')
|
||||
assert.equal(b.holdings().has('u1'), false, '失败不得写入本地持有记录')
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: 续租必须带 epoch —— 旧持有者续租失败(fencing 生效)', async () => {
|
||||
const db = await freshDb()
|
||||
const a = shortLease(db, 'w-a')
|
||||
assert.equal((await a.acquire('u1')).ok, true)
|
||||
const staleEpoch = a.holdings().get('u1').epoch
|
||||
|
||||
await sleep(90) // 让租约过期
|
||||
const b = shortLease(db, 'w-b')
|
||||
const taken = await b.acquire('u1')
|
||||
assert.equal(taken.ok, true, '过期后可被抢占')
|
||||
assert.equal(taken.epoch, staleEpoch + 1, 'epoch 必须递增')
|
||||
|
||||
// 老持有者拿着旧 epoch 续租 ⇒ 必须失败(否则就脑裂双写了)
|
||||
assert.equal(await a.renew('u1'), false)
|
||||
assert.equal(a.holdings().has('u1'), false, '失权后必须清掉本地记录(供 self-fence)')
|
||||
|
||||
// 直接调 DB 层也一样:epoch 不匹配 → 不更新
|
||||
assert.equal(await db.renewInstanceLease('u1', 'w-a', staleEpoch, 60), false)
|
||||
assert.equal(await b.renew('u1'), true, '新持有者续租成功')
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: 释放后归零,可再次抢占且 epoch 继续递增', async () => {
|
||||
const db = await freshDb()
|
||||
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
|
||||
await a.acquire('u1')
|
||||
assert.equal(await a.release('u1'), true)
|
||||
assert.equal(a.holdings().has('u1'), false)
|
||||
|
||||
const inst = await db.findUserInstance('u1', 'main')
|
||||
assert.equal(inst.hostId, null, '释放 = 归属清空')
|
||||
assert.equal(stillHolder(inst, 'w-a', 1), false)
|
||||
|
||||
const b = new InstanceLease(db, 'w-b', { ttlMs: 5000, renewMs: 1000 })
|
||||
const r = await b.acquire('u1')
|
||||
assert.equal(r.ok, true)
|
||||
assert.equal(r.epoch, 2, 'epoch 单调递增(不复用)')
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: release 也带 epoch 校验 —— 老持有者不能清掉新持有者的归属', async () => {
|
||||
const db = await freshDb()
|
||||
const a = shortLease(db, 'w-a')
|
||||
await a.acquire('u1')
|
||||
await sleep(90)
|
||||
const b = shortLease(db, 'w-b')
|
||||
await b.acquire('u1')
|
||||
|
||||
// a 试图释放(它本地已失权 ⇒ release 返回 false 且不动 DB)
|
||||
assert.equal(await a.release('u1'), false)
|
||||
const inst = await db.findUserInstance('u1', 'main')
|
||||
assert.equal(inst.hostId, 'w-b', '新持有者的归属不能被误清')
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: stillHolder 判据(hostId + epoch 双匹配)', async () => {
|
||||
const db = await freshDb()
|
||||
const a = new InstanceLease(db, 'w-a', { ttlMs: 5000, renewMs: 1000 })
|
||||
await a.acquire('u1')
|
||||
const inst = await db.findUserInstance('u1', 'main')
|
||||
|
||||
assert.equal(stillHolder(inst, 'w-a', 1), true)
|
||||
assert.equal(stillHolder(inst, 'w-a', 2), false, 'epoch 不符 = 已失权')
|
||||
assert.equal(stillHolder(inst, 'w-b', 1), false, '换了机器 = 已失权')
|
||||
assert.equal(stillHolder(undefined, 'w-a', 1), false)
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: 对账视图 —— mine() 只回本机、expiredAll() 回全局过期', async () => {
|
||||
const db = await freshDb()
|
||||
await db.createUser({ id: 'u2', username: 'bob', passHash: 'x', role: 'active', homeDir: '/tmp/u2' })
|
||||
const a = shortLease(db, 'w-a')
|
||||
const b = shortLease(db, 'w-b')
|
||||
await a.acquire('u1')
|
||||
await b.acquire('u2')
|
||||
|
||||
assert.deepEqual((await a.mine()).map((i) => i.userId), ['u1'], '一次拿回整机(本机只有 u1)')
|
||||
assert.deepEqual((await b.mine()).map((i) => i.userId), ['u2'])
|
||||
assert.equal((await a.expiredAll()).length, 0, '刚认领未过期')
|
||||
|
||||
await sleep(90)
|
||||
assert.equal((await a.expiredAll()).length, 2, '过期后两台都进清单(供巡检,不自动接管)')
|
||||
assert.equal((await a.expiredHere()).length, 1, 'expiredHere 只回自己名下')
|
||||
await db.close()
|
||||
})
|
||||
|
||||
test('lease: renewAll 回传失权清单(调用方据此 self-fence)', async () => {
|
||||
const db = await freshDb()
|
||||
const a = shortLease(db, 'w-a')
|
||||
await a.acquire('u1')
|
||||
assert.deepEqual(await a.renewAll(), [], '正常时无人失权')
|
||||
|
||||
await sleep(90)
|
||||
const b = shortLease(db, 'w-b')
|
||||
await b.acquire('u1') // 抢走
|
||||
assert.deepEqual(await a.renewAll(), ['u1'], 'a 必须知道自己已失权')
|
||||
await db.close()
|
||||
})
|
||||
Reference in new issue
Block a user